Archive stopped generator: latest 45k, best validation 44k, results and best Gen-PPL provenance
Browse files- iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-44000.pt +3 -0
- iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-45000.pt +3 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/EVAL_DONE.json +5 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.samples.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.samples.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.samples.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.samples.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.samples.json +0 -0
- iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/summary.json +35 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp-config.yaml +179 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp.pt +3 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae-config.yaml +117 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae.pt +3 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp-config.yaml +179 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp.pt +3 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae-config.yaml +117 -0
- iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae.pt +3 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/README.md +7 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/STOPPED.json +10 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/STOP_REQUESTED.json +21 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/backup-manifest.json +244 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/config.yaml +223 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/git-provenance.json +3 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/manifest.json +756 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/progress.json +14 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-12875.log +0 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-25750.log +0 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-4769.log +0 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-51499.log +0 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-954.log +185 -0
- iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/validation-curve.json +492 -0
iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-44000.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dd84df67dc533dcd4e3554e83363275b247d59abfa8c06d0936343eea2277f31
|
| 3 |
+
size 2565487307
|
iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-45000.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:729d7432a0d99e7e1ccacd1526ffe0426e374df87c9037da52a0af0a60dee726
|
| 3 |
+
size 2565487243
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/EVAL_DONE.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 25750,
|
| 3 |
+
"reference_dtype": "float32",
|
| 4 |
+
"sampling": "supplied-level0-untruncated"
|
| 5 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.samples.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.samples.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.samples.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.samples.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.samples.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/summary.json
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"protocol": {
|
| 3 |
+
"sampling": "random",
|
| 4 |
+
"temperature": 1.0,
|
| 5 |
+
"top_k": 0,
|
| 6 |
+
"top_p": 1.0,
|
| 7 |
+
"n_provided_levels": 1,
|
| 8 |
+
"n_levels": 16,
|
| 9 |
+
"document_aware_validation_sampling": true
|
| 10 |
+
},
|
| 11 |
+
"reference_model": "gpt2-large",
|
| 12 |
+
"checkpoint": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-25750/ncp.pt",
|
| 13 |
+
"checkpoint_step": 25750,
|
| 14 |
+
"seeds": [
|
| 15 |
+
0,
|
| 16 |
+
1,
|
| 17 |
+
2,
|
| 18 |
+
3,
|
| 19 |
+
4
|
| 20 |
+
],
|
| 21 |
+
"samples_per_seed": 128,
|
| 22 |
+
"total_scored_samples": 640,
|
| 23 |
+
"total_scored_tokens": 160504,
|
| 24 |
+
"per_seed_mean_ppl": {
|
| 25 |
+
"0": 205.27390254957734,
|
| 26 |
+
"1": 191.09995809495,
|
| 27 |
+
"2": 199.86937893942272,
|
| 28 |
+
"3": 185.8896512015558,
|
| 29 |
+
"4": 168.47801285382266
|
| 30 |
+
},
|
| 31 |
+
"mean_gen_ppl": 190.1221807278657,
|
| 32 |
+
"gen_ppl_se": 6.371510436577567,
|
| 33 |
+
"mean_of_seed_median_ppl": 179.17403564453124,
|
| 34 |
+
"mean_entropy_nats": 4.466862099671543
|
| 35 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp-config.yaml
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: next-concept
|
| 2 |
+
experiment: iclr-debug-two-stage-135b-20260916-generator
|
| 3 |
+
experiment_dir: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000
|
| 4 |
+
model:
|
| 5 |
+
n_layer: 12
|
| 6 |
+
n_head: 12
|
| 7 |
+
bias: true
|
| 8 |
+
dropout: 0.1
|
| 9 |
+
n_embd: 768
|
| 10 |
+
context_length: 256
|
| 11 |
+
vocab_size:
|
| 12 |
+
- 16384
|
| 13 |
+
- 16384
|
| 14 |
+
- 16384
|
| 15 |
+
- 16384
|
| 16 |
+
- 16384
|
| 17 |
+
- 16384
|
| 18 |
+
- 16384
|
| 19 |
+
- 16384
|
| 20 |
+
- 16384
|
| 21 |
+
- 16384
|
| 22 |
+
- 16384
|
| 23 |
+
- 16384
|
| 24 |
+
- 16384
|
| 25 |
+
- 16384
|
| 26 |
+
- 16384
|
| 27 |
+
attn_pattern: block_diagonal
|
| 28 |
+
use_positional_encoding: true
|
| 29 |
+
use_level_encoding: true
|
| 30 |
+
prefix_len: 0
|
| 31 |
+
use_rope: true
|
| 32 |
+
rope_base: 10000
|
| 33 |
+
use_qk_norm: true
|
| 34 |
+
post_upsample_conv:
|
| 35 |
+
enabled: true
|
| 36 |
+
kernel_size: 3
|
| 37 |
+
use_relu2: true
|
| 38 |
+
use_flex_attention: true
|
| 39 |
+
shared_output_head: true
|
| 40 |
+
shared_head_per_level_bias: false
|
| 41 |
+
shared_head_per_level_scale: false
|
| 42 |
+
shared_head_adapter_rank: 0
|
| 43 |
+
tokenizer:
|
| 44 |
+
pre_tokenizer:
|
| 45 |
+
name: hf
|
| 46 |
+
tokenizer: hf
|
| 47 |
+
model_id: gpt2
|
| 48 |
+
special_tokens:
|
| 49 |
+
eos_token: <|endoftext|>
|
| 50 |
+
pad_token: <|pad|>
|
| 51 |
+
vqvae:
|
| 52 |
+
n_layers: 6
|
| 53 |
+
context_length: 256
|
| 54 |
+
embed_dim: 384
|
| 55 |
+
in_vocab_size: 50304
|
| 56 |
+
vqvae_vocab_size:
|
| 57 |
+
- 16384
|
| 58 |
+
- 16384
|
| 59 |
+
- 16384
|
| 60 |
+
- 16384
|
| 61 |
+
- 16384
|
| 62 |
+
- 16384
|
| 63 |
+
- 16384
|
| 64 |
+
- 16384
|
| 65 |
+
- 16384
|
| 66 |
+
- 16384
|
| 67 |
+
- 16384
|
| 68 |
+
- 16384
|
| 69 |
+
- 16384
|
| 70 |
+
- 16384
|
| 71 |
+
- 16384
|
| 72 |
+
- 16384
|
| 73 |
+
compression_factor: 1
|
| 74 |
+
dropout: 0.1
|
| 75 |
+
pre_quant_groupnorm: 4
|
| 76 |
+
pre_quant_dropout: 0.2
|
| 77 |
+
vector_quantizer_config:
|
| 78 |
+
clss: multiscale_residual_vector_quantizer
|
| 79 |
+
decay: 0.99
|
| 80 |
+
epsilon: 1.0e-05
|
| 81 |
+
commitment_cost: 0.25
|
| 82 |
+
learned_l1_sampling: true
|
| 83 |
+
learned_all_sampling: false
|
| 84 |
+
quant_resi:
|
| 85 |
+
enabled: true
|
| 86 |
+
ratio: 0.5
|
| 87 |
+
share_mode: 0
|
| 88 |
+
learnable_ratio: false
|
| 89 |
+
levels:
|
| 90 |
+
use_manual_levels: true
|
| 91 |
+
manual_levels:
|
| 92 |
+
- 1
|
| 93 |
+
- 4
|
| 94 |
+
- 9
|
| 95 |
+
- 16
|
| 96 |
+
- 25
|
| 97 |
+
- 36
|
| 98 |
+
- 49
|
| 99 |
+
- 64
|
| 100 |
+
- 81
|
| 101 |
+
- 100
|
| 102 |
+
- 121
|
| 103 |
+
- 144
|
| 104 |
+
- 169
|
| 105 |
+
- 196
|
| 106 |
+
- 225
|
| 107 |
+
- 256
|
| 108 |
+
aux:
|
| 109 |
+
fine_drop:
|
| 110 |
+
prob: 0.5
|
| 111 |
+
min_keep: 1
|
| 112 |
+
checkpoint_path: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae.pt
|
| 113 |
+
dataset:
|
| 114 |
+
train_source: /home/ubuntu/data/full_owt/train_gpt2.bin
|
| 115 |
+
validate_source: /home/ubuntu/data/full_owt/valid_gpt2.bin
|
| 116 |
+
pad_token_id: 50257
|
| 117 |
+
wandb:
|
| 118 |
+
entity: mstok
|
| 119 |
+
project: owt-repro
|
| 120 |
+
group: owtsmall-repro-ctx256-sharedhead-ncp
|
| 121 |
+
enable: false
|
| 122 |
+
id: owtsmall-repro-ctx256-sharedhead-ncp-5ep
|
| 123 |
+
resume: allow
|
| 124 |
+
gradients_and_params:
|
| 125 |
+
enable: false
|
| 126 |
+
log: all
|
| 127 |
+
log_freq: 1000
|
| 128 |
+
optimization:
|
| 129 |
+
lr: 0.0005
|
| 130 |
+
reset_lr: true
|
| 131 |
+
warmup_iters: 300
|
| 132 |
+
lr_decay_iters: null
|
| 133 |
+
min_lr: 1.0e-05
|
| 134 |
+
beta_1: 0.9
|
| 135 |
+
beta_2: 0.99
|
| 136 |
+
weight_decay: 0.1
|
| 137 |
+
max_grad_norm: 1.0
|
| 138 |
+
grad_accumulation_steps: 4
|
| 139 |
+
corruption:
|
| 140 |
+
mode: per_level
|
| 141 |
+
per_level_probs:
|
| 142 |
+
- 0.85
|
| 143 |
+
- 0.8321428571
|
| 144 |
+
- 0.8142857143
|
| 145 |
+
- 0.7964285714
|
| 146 |
+
- 0.7785714286
|
| 147 |
+
- 0.7607142857
|
| 148 |
+
- 0.7428571429
|
| 149 |
+
- 0.725
|
| 150 |
+
- 0.7071428571
|
| 151 |
+
- 0.6892857143
|
| 152 |
+
- 0.6714285714
|
| 153 |
+
- 0.6535714286
|
| 154 |
+
- 0.6357142857
|
| 155 |
+
- 0.6178571429
|
| 156 |
+
- 0.6
|
| 157 |
+
skip_level0: true
|
| 158 |
+
level_loss_alpha: 1.0
|
| 159 |
+
training:
|
| 160 |
+
seed: 55
|
| 161 |
+
log_interval: 10
|
| 162 |
+
eval_interval: 5000
|
| 163 |
+
checkpoint_interval: 2500
|
| 164 |
+
val_iters: 100
|
| 165 |
+
total_iters: null
|
| 166 |
+
tqdm_interval: 10
|
| 167 |
+
n_epochs: 5
|
| 168 |
+
batch_size: 96
|
| 169 |
+
eval_batch_size: 96
|
| 170 |
+
demo_batch_size: 4
|
| 171 |
+
keep_last: 8
|
| 172 |
+
torch_compile:
|
| 173 |
+
enable: false
|
| 174 |
+
mode: default
|
| 175 |
+
dynamic: false
|
| 176 |
+
fullgraph: false
|
| 177 |
+
backend: inductor
|
| 178 |
+
mixed_precision:
|
| 179 |
+
enable: true
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:59e5405c493f4c5310dd9f9c53e3a5563de8902b387c281bb079dfa347a5c88d
|
| 3 |
+
size 427469662
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae-config.yaml
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: summarization
|
| 2 |
+
experiment: iclr-debug-two-stage-135b-20260916-generator
|
| 3 |
+
experiment_dir: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000
|
| 4 |
+
training:
|
| 5 |
+
seed: 42
|
| 6 |
+
log_interval: 10
|
| 7 |
+
eval_interval: 10000
|
| 8 |
+
checkpoint_interval: 10000
|
| 9 |
+
val_iters: 100
|
| 10 |
+
n_epochs: 10
|
| 11 |
+
total_iters: null
|
| 12 |
+
tqdm_interval: 10
|
| 13 |
+
batch_size: 256
|
| 14 |
+
eval_batch_size: 128
|
| 15 |
+
demo_batch_size: 4
|
| 16 |
+
model:
|
| 17 |
+
n_layers: 6
|
| 18 |
+
context_length: 256
|
| 19 |
+
embed_dim: 384
|
| 20 |
+
in_vocab_size: 50304
|
| 21 |
+
vqvae_vocab_size:
|
| 22 |
+
- 16384
|
| 23 |
+
- 16384
|
| 24 |
+
- 16384
|
| 25 |
+
- 16384
|
| 26 |
+
- 16384
|
| 27 |
+
- 16384
|
| 28 |
+
- 16384
|
| 29 |
+
- 16384
|
| 30 |
+
- 16384
|
| 31 |
+
- 16384
|
| 32 |
+
- 16384
|
| 33 |
+
- 16384
|
| 34 |
+
- 16384
|
| 35 |
+
- 16384
|
| 36 |
+
- 16384
|
| 37 |
+
- 16384
|
| 38 |
+
compression_factor: 1
|
| 39 |
+
dropout: 0.1
|
| 40 |
+
pre_quant_groupnorm: 4
|
| 41 |
+
pre_quant_dropout: 0.2
|
| 42 |
+
vector_quantizer:
|
| 43 |
+
clss: multiscale_residual_vector_quantizer
|
| 44 |
+
decay: 0.99
|
| 45 |
+
epsilon: 1.0e-05
|
| 46 |
+
commitment_cost: 0.25
|
| 47 |
+
learned_l1_sampling: true
|
| 48 |
+
learned_all_sampling: false
|
| 49 |
+
quant_resi:
|
| 50 |
+
enabled: true
|
| 51 |
+
ratio: 0.5
|
| 52 |
+
share_mode: 0
|
| 53 |
+
learnable_ratio: false
|
| 54 |
+
levels:
|
| 55 |
+
use_manual_levels: true
|
| 56 |
+
manual_levels:
|
| 57 |
+
- 1
|
| 58 |
+
- 4
|
| 59 |
+
- 9
|
| 60 |
+
- 16
|
| 61 |
+
- 25
|
| 62 |
+
- 36
|
| 63 |
+
- 49
|
| 64 |
+
- 64
|
| 65 |
+
- 81
|
| 66 |
+
- 100
|
| 67 |
+
- 121
|
| 68 |
+
- 144
|
| 69 |
+
- 169
|
| 70 |
+
- 196
|
| 71 |
+
- 225
|
| 72 |
+
- 256
|
| 73 |
+
aux:
|
| 74 |
+
fine_drop:
|
| 75 |
+
prob: 0.5
|
| 76 |
+
min_keep: 1
|
| 77 |
+
wandb:
|
| 78 |
+
entity: mstok
|
| 79 |
+
project: owt-repro
|
| 80 |
+
enable: false
|
| 81 |
+
group: owtsmall-repro-ctx256-vqvae
|
| 82 |
+
id: owtsmall-repro-ctx256-vqvae-10ep
|
| 83 |
+
resume: allow
|
| 84 |
+
gradients_and_params:
|
| 85 |
+
enable: false
|
| 86 |
+
log: all
|
| 87 |
+
log_freq: 1000
|
| 88 |
+
optimization:
|
| 89 |
+
lr: 0.001
|
| 90 |
+
lr_decay_iters: null
|
| 91 |
+
min_lr: null
|
| 92 |
+
beta_1: 0.9
|
| 93 |
+
beta_2: 0.99
|
| 94 |
+
weight_decay: 0.1
|
| 95 |
+
max_grad_norm: 1.0
|
| 96 |
+
grad_accumulation_steps: 2
|
| 97 |
+
torch_compile:
|
| 98 |
+
enable: false
|
| 99 |
+
mode: default
|
| 100 |
+
dynamic: false
|
| 101 |
+
fullgraph: false
|
| 102 |
+
backend: inductor
|
| 103 |
+
mixed_precision:
|
| 104 |
+
enable: true
|
| 105 |
+
opt_level: O1
|
| 106 |
+
loss_scale: dynamic
|
| 107 |
+
dataset:
|
| 108 |
+
train_source: /home/ubuntu/data/full_owt/train_gpt2.bin
|
| 109 |
+
validate_source: /home/ubuntu/data/full_owt/valid_gpt2.bin
|
| 110 |
+
pad_token_id: 50257
|
| 111 |
+
pre_tokenizer:
|
| 112 |
+
name: hf
|
| 113 |
+
tokenizer: hf
|
| 114 |
+
model_id: gpt2
|
| 115 |
+
special_tokens:
|
| 116 |
+
eos_token: <|endoftext|>
|
| 117 |
+
pad_token: <|pad|>
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f85bbb9c51879b97f5dfa175815335643d2c18169568455e88731b84715fcdab
|
| 3 |
+
size 1300864085
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp-config.yaml
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: next-concept
|
| 2 |
+
experiment: iclr-debug-two-stage-135b-20260916-generator
|
| 3 |
+
experiment_dir: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000
|
| 4 |
+
model:
|
| 5 |
+
n_layer: 12
|
| 6 |
+
n_head: 12
|
| 7 |
+
bias: true
|
| 8 |
+
dropout: 0.1
|
| 9 |
+
n_embd: 768
|
| 10 |
+
context_length: 256
|
| 11 |
+
vocab_size:
|
| 12 |
+
- 16384
|
| 13 |
+
- 16384
|
| 14 |
+
- 16384
|
| 15 |
+
- 16384
|
| 16 |
+
- 16384
|
| 17 |
+
- 16384
|
| 18 |
+
- 16384
|
| 19 |
+
- 16384
|
| 20 |
+
- 16384
|
| 21 |
+
- 16384
|
| 22 |
+
- 16384
|
| 23 |
+
- 16384
|
| 24 |
+
- 16384
|
| 25 |
+
- 16384
|
| 26 |
+
- 16384
|
| 27 |
+
attn_pattern: block_diagonal
|
| 28 |
+
use_positional_encoding: true
|
| 29 |
+
use_level_encoding: true
|
| 30 |
+
prefix_len: 0
|
| 31 |
+
use_rope: true
|
| 32 |
+
rope_base: 10000
|
| 33 |
+
use_qk_norm: true
|
| 34 |
+
post_upsample_conv:
|
| 35 |
+
enabled: true
|
| 36 |
+
kernel_size: 3
|
| 37 |
+
use_relu2: true
|
| 38 |
+
use_flex_attention: true
|
| 39 |
+
shared_output_head: true
|
| 40 |
+
shared_head_per_level_bias: false
|
| 41 |
+
shared_head_per_level_scale: false
|
| 42 |
+
shared_head_adapter_rank: 0
|
| 43 |
+
tokenizer:
|
| 44 |
+
pre_tokenizer:
|
| 45 |
+
name: hf
|
| 46 |
+
tokenizer: hf
|
| 47 |
+
model_id: gpt2
|
| 48 |
+
special_tokens:
|
| 49 |
+
eos_token: <|endoftext|>
|
| 50 |
+
pad_token: <|pad|>
|
| 51 |
+
vqvae:
|
| 52 |
+
n_layers: 6
|
| 53 |
+
context_length: 256
|
| 54 |
+
embed_dim: 384
|
| 55 |
+
in_vocab_size: 50304
|
| 56 |
+
vqvae_vocab_size:
|
| 57 |
+
- 16384
|
| 58 |
+
- 16384
|
| 59 |
+
- 16384
|
| 60 |
+
- 16384
|
| 61 |
+
- 16384
|
| 62 |
+
- 16384
|
| 63 |
+
- 16384
|
| 64 |
+
- 16384
|
| 65 |
+
- 16384
|
| 66 |
+
- 16384
|
| 67 |
+
- 16384
|
| 68 |
+
- 16384
|
| 69 |
+
- 16384
|
| 70 |
+
- 16384
|
| 71 |
+
- 16384
|
| 72 |
+
- 16384
|
| 73 |
+
compression_factor: 1
|
| 74 |
+
dropout: 0.1
|
| 75 |
+
pre_quant_groupnorm: 4
|
| 76 |
+
pre_quant_dropout: 0.2
|
| 77 |
+
vector_quantizer_config:
|
| 78 |
+
clss: multiscale_residual_vector_quantizer
|
| 79 |
+
decay: 0.99
|
| 80 |
+
epsilon: 1.0e-05
|
| 81 |
+
commitment_cost: 0.25
|
| 82 |
+
learned_l1_sampling: true
|
| 83 |
+
learned_all_sampling: false
|
| 84 |
+
quant_resi:
|
| 85 |
+
enabled: true
|
| 86 |
+
ratio: 0.5
|
| 87 |
+
share_mode: 0
|
| 88 |
+
learnable_ratio: false
|
| 89 |
+
levels:
|
| 90 |
+
use_manual_levels: true
|
| 91 |
+
manual_levels:
|
| 92 |
+
- 1
|
| 93 |
+
- 4
|
| 94 |
+
- 9
|
| 95 |
+
- 16
|
| 96 |
+
- 25
|
| 97 |
+
- 36
|
| 98 |
+
- 49
|
| 99 |
+
- 64
|
| 100 |
+
- 81
|
| 101 |
+
- 100
|
| 102 |
+
- 121
|
| 103 |
+
- 144
|
| 104 |
+
- 169
|
| 105 |
+
- 196
|
| 106 |
+
- 225
|
| 107 |
+
- 256
|
| 108 |
+
aux:
|
| 109 |
+
fine_drop:
|
| 110 |
+
prob: 0.5
|
| 111 |
+
min_keep: 1
|
| 112 |
+
checkpoint_path: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae.pt
|
| 113 |
+
dataset:
|
| 114 |
+
train_source: /home/ubuntu/data/full_owt/train_gpt2.bin
|
| 115 |
+
validate_source: /home/ubuntu/data/full_owt/valid_gpt2.bin
|
| 116 |
+
pad_token_id: 50257
|
| 117 |
+
wandb:
|
| 118 |
+
entity: mstok
|
| 119 |
+
project: owt-repro
|
| 120 |
+
group: owtsmall-repro-ctx256-sharedhead-ncp
|
| 121 |
+
enable: false
|
| 122 |
+
id: owtsmall-repro-ctx256-sharedhead-ncp-5ep
|
| 123 |
+
resume: allow
|
| 124 |
+
gradients_and_params:
|
| 125 |
+
enable: false
|
| 126 |
+
log: all
|
| 127 |
+
log_freq: 1000
|
| 128 |
+
optimization:
|
| 129 |
+
lr: 0.0005
|
| 130 |
+
reset_lr: true
|
| 131 |
+
warmup_iters: 300
|
| 132 |
+
lr_decay_iters: null
|
| 133 |
+
min_lr: 1.0e-05
|
| 134 |
+
beta_1: 0.9
|
| 135 |
+
beta_2: 0.99
|
| 136 |
+
weight_decay: 0.1
|
| 137 |
+
max_grad_norm: 1.0
|
| 138 |
+
grad_accumulation_steps: 4
|
| 139 |
+
corruption:
|
| 140 |
+
mode: per_level
|
| 141 |
+
per_level_probs:
|
| 142 |
+
- 0.85
|
| 143 |
+
- 0.8321428571
|
| 144 |
+
- 0.8142857143
|
| 145 |
+
- 0.7964285714
|
| 146 |
+
- 0.7785714286
|
| 147 |
+
- 0.7607142857
|
| 148 |
+
- 0.7428571429
|
| 149 |
+
- 0.725
|
| 150 |
+
- 0.7071428571
|
| 151 |
+
- 0.6892857143
|
| 152 |
+
- 0.6714285714
|
| 153 |
+
- 0.6535714286
|
| 154 |
+
- 0.6357142857
|
| 155 |
+
- 0.6178571429
|
| 156 |
+
- 0.6
|
| 157 |
+
skip_level0: true
|
| 158 |
+
level_loss_alpha: 1.0
|
| 159 |
+
training:
|
| 160 |
+
seed: 55
|
| 161 |
+
log_interval: 10
|
| 162 |
+
eval_interval: 5000
|
| 163 |
+
checkpoint_interval: 2500
|
| 164 |
+
val_iters: 100
|
| 165 |
+
total_iters: null
|
| 166 |
+
tqdm_interval: 10
|
| 167 |
+
n_epochs: 5
|
| 168 |
+
batch_size: 96
|
| 169 |
+
eval_batch_size: 96
|
| 170 |
+
demo_batch_size: 4
|
| 171 |
+
keep_last: 8
|
| 172 |
+
torch_compile:
|
| 173 |
+
enable: false
|
| 174 |
+
mode: default
|
| 175 |
+
dynamic: false
|
| 176 |
+
fullgraph: false
|
| 177 |
+
backend: inductor
|
| 178 |
+
mixed_precision:
|
| 179 |
+
enable: true
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ed0cf1f72beda0fc582af91014fd1dc7dcc8a2a23c7f38b91c853020da1db04a
|
| 3 |
+
size 427469662
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae-config.yaml
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: summarization
|
| 2 |
+
experiment: iclr-debug-two-stage-135b-20260916-generator
|
| 3 |
+
experiment_dir: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000
|
| 4 |
+
training:
|
| 5 |
+
seed: 42
|
| 6 |
+
log_interval: 10
|
| 7 |
+
eval_interval: 10000
|
| 8 |
+
checkpoint_interval: 10000
|
| 9 |
+
val_iters: 100
|
| 10 |
+
n_epochs: 10
|
| 11 |
+
total_iters: null
|
| 12 |
+
tqdm_interval: 10
|
| 13 |
+
batch_size: 256
|
| 14 |
+
eval_batch_size: 128
|
| 15 |
+
demo_batch_size: 4
|
| 16 |
+
model:
|
| 17 |
+
n_layers: 6
|
| 18 |
+
context_length: 256
|
| 19 |
+
embed_dim: 384
|
| 20 |
+
in_vocab_size: 50304
|
| 21 |
+
vqvae_vocab_size:
|
| 22 |
+
- 16384
|
| 23 |
+
- 16384
|
| 24 |
+
- 16384
|
| 25 |
+
- 16384
|
| 26 |
+
- 16384
|
| 27 |
+
- 16384
|
| 28 |
+
- 16384
|
| 29 |
+
- 16384
|
| 30 |
+
- 16384
|
| 31 |
+
- 16384
|
| 32 |
+
- 16384
|
| 33 |
+
- 16384
|
| 34 |
+
- 16384
|
| 35 |
+
- 16384
|
| 36 |
+
- 16384
|
| 37 |
+
- 16384
|
| 38 |
+
compression_factor: 1
|
| 39 |
+
dropout: 0.1
|
| 40 |
+
pre_quant_groupnorm: 4
|
| 41 |
+
pre_quant_dropout: 0.2
|
| 42 |
+
vector_quantizer:
|
| 43 |
+
clss: multiscale_residual_vector_quantizer
|
| 44 |
+
decay: 0.99
|
| 45 |
+
epsilon: 1.0e-05
|
| 46 |
+
commitment_cost: 0.25
|
| 47 |
+
learned_l1_sampling: true
|
| 48 |
+
learned_all_sampling: false
|
| 49 |
+
quant_resi:
|
| 50 |
+
enabled: true
|
| 51 |
+
ratio: 0.5
|
| 52 |
+
share_mode: 0
|
| 53 |
+
learnable_ratio: false
|
| 54 |
+
levels:
|
| 55 |
+
use_manual_levels: true
|
| 56 |
+
manual_levels:
|
| 57 |
+
- 1
|
| 58 |
+
- 4
|
| 59 |
+
- 9
|
| 60 |
+
- 16
|
| 61 |
+
- 25
|
| 62 |
+
- 36
|
| 63 |
+
- 49
|
| 64 |
+
- 64
|
| 65 |
+
- 81
|
| 66 |
+
- 100
|
| 67 |
+
- 121
|
| 68 |
+
- 144
|
| 69 |
+
- 169
|
| 70 |
+
- 196
|
| 71 |
+
- 225
|
| 72 |
+
- 256
|
| 73 |
+
aux:
|
| 74 |
+
fine_drop:
|
| 75 |
+
prob: 0.5
|
| 76 |
+
min_keep: 1
|
| 77 |
+
wandb:
|
| 78 |
+
entity: mstok
|
| 79 |
+
project: owt-repro
|
| 80 |
+
enable: false
|
| 81 |
+
group: owtsmall-repro-ctx256-vqvae
|
| 82 |
+
id: owtsmall-repro-ctx256-vqvae-10ep
|
| 83 |
+
resume: allow
|
| 84 |
+
gradients_and_params:
|
| 85 |
+
enable: false
|
| 86 |
+
log: all
|
| 87 |
+
log_freq: 1000
|
| 88 |
+
optimization:
|
| 89 |
+
lr: 0.001
|
| 90 |
+
lr_decay_iters: null
|
| 91 |
+
min_lr: null
|
| 92 |
+
beta_1: 0.9
|
| 93 |
+
beta_2: 0.99
|
| 94 |
+
weight_decay: 0.1
|
| 95 |
+
max_grad_norm: 1.0
|
| 96 |
+
grad_accumulation_steps: 2
|
| 97 |
+
torch_compile:
|
| 98 |
+
enable: false
|
| 99 |
+
mode: default
|
| 100 |
+
dynamic: false
|
| 101 |
+
fullgraph: false
|
| 102 |
+
backend: inductor
|
| 103 |
+
mixed_precision:
|
| 104 |
+
enable: true
|
| 105 |
+
opt_level: O1
|
| 106 |
+
loss_scale: dynamic
|
| 107 |
+
dataset:
|
| 108 |
+
train_source: /home/ubuntu/data/full_owt/train_gpt2.bin
|
| 109 |
+
validate_source: /home/ubuntu/data/full_owt/valid_gpt2.bin
|
| 110 |
+
pad_token_id: 50257
|
| 111 |
+
pre_tokenizer:
|
| 112 |
+
name: hf
|
| 113 |
+
tokenizer: hf
|
| 114 |
+
model_id: gpt2
|
| 115 |
+
special_tokens:
|
| 116 |
+
eos_token: <|endoftext|>
|
| 117 |
+
pad_token: <|pad|>
|
iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a16ecf6a1b8fb925e65bbcc120fce5fee7f0705e882c25351a1c4a6f6be0ceba
|
| 3 |
+
size 1300864085
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/README.md
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Stopped original generator run — September 17, 2026
|
| 2 |
+
|
| 3 |
+
Stopped at user request to test a fresh generator with faster learning-rate decay. The original training schedule decayed over 135B positions. Latest retained resumable checkpoint: step 45,000 (47.186B positions). Best validation-loss checkpoint: step 44,000. Best measured generation-PPL checkpoint: step 12,875 (13.500B positions), 174.9645 ± 3.4274 seed SE under strict FP32 GPT-2 Large scoring, five seeds × 128 samples. This checkpoint and its full results were previously backed up and are verified again in this archive.
|
| 4 |
+
|
| 5 |
+
Step 22,000 scored 189.300829 in FP32 and 189.300857 in FP64 on identical samples; it remains available with its checkpoint and precision audit. The scheduled step 25,750 evaluation scored 190.1222 ± 6.3715 but had TF32 enabled; it is not a strict FP32/FP64 comparison. No generation-PPL claim is made for steps 44,000 or 45,000.
|
| 6 |
+
|
| 7 |
+
All generators here use the frozen final tokenizer from step 128,747. The restart uses that exact tokenizer again, fresh generator and optimizer state, the same initialization seeds, and a 27B-position cosine horizon followed by minimum LR to 135B. Samples remain level-0 conditioned, untruncated temperature-1, top-k 0, top-p 1. Full training states include optimizer, scheduler, token count, and per-rank RNG state. See the archive manifest for SHA-256 verification and validation-curve.json for all validation measurements.
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/STOPPED.json
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"stage": "stopped-by-user",
|
| 3 |
+
"updated_utc": "2026-09-17T22:25:15.271975+00:00",
|
| 4 |
+
"last_logged_step": 45320,
|
| 5 |
+
"latest_resumable_step": 45000,
|
| 6 |
+
"best_validation_step": 44000,
|
| 7 |
+
"best_measured_gen_ppl_step": 12875,
|
| 8 |
+
"best_measured_gen_ppl": 174.96447636439655,
|
| 9 |
+
"reason": "Fresh generator with same final tokenizer, cosine decay over first 27B positions then constant minimum LR to 135B"
|
| 10 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/STOP_REQUESTED.json
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"requested_utc": "2026-09-17T22:23:44.577915+00:00",
|
| 3 |
+
"reason": "User requested stopping and fresh generator with LR decay over 27B then minimum LR to 135B",
|
| 4 |
+
"retained_checkpoints": [
|
| 5 |
+
{
|
| 6 |
+
"source": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-45000.pt",
|
| 7 |
+
"step": 45000,
|
| 8 |
+
"retained": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/manual-checkpoints/checkpoint-iter-45000.pt"
|
| 9 |
+
},
|
| 10 |
+
{
|
| 11 |
+
"source": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/best_checkpoint.pt",
|
| 12 |
+
"step": 44000,
|
| 13 |
+
"retained": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/manual-checkpoints/checkpoint-iter-44000.pt"
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"source": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/milestone-iter-12875.pt",
|
| 17 |
+
"step": 12875,
|
| 18 |
+
"retained": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/manual-checkpoints/checkpoint-iter-12875.pt"
|
| 19 |
+
}
|
| 20 |
+
]
|
| 21 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/backup-manifest.json
ADDED
|
@@ -0,0 +1,244 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"repo": "iskhare/iclr-debug",
|
| 3 |
+
"run": "iclr-debug-two-stage-135b-20260916",
|
| 4 |
+
"stopped": {
|
| 5 |
+
"stage": "stopped-by-user",
|
| 6 |
+
"updated_utc": "2026-09-17T22:25:15.271975+00:00",
|
| 7 |
+
"last_logged_step": 45320,
|
| 8 |
+
"latest_resumable_step": 45000,
|
| 9 |
+
"best_validation_step": 44000,
|
| 10 |
+
"best_measured_gen_ppl_step": 12875,
|
| 11 |
+
"best_measured_gen_ppl": 174.96447636439655,
|
| 12 |
+
"reason": "Fresh generator with same final tokenizer, cosine decay over first 27B positions then constant minimum LR to 135B"
|
| 13 |
+
},
|
| 14 |
+
"files": [
|
| 15 |
+
{
|
| 16 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/manual-checkpoints/checkpoint-iter-44000.pt",
|
| 17 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-44000.pt",
|
| 18 |
+
"size_bytes": 2565487307,
|
| 19 |
+
"sha256": "dd84df67dc533dcd4e3554e83363275b247d59abfa8c06d0936343eea2277f31"
|
| 20 |
+
},
|
| 21 |
+
{
|
| 22 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp-config.yaml",
|
| 23 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp-config.yaml",
|
| 24 |
+
"size_bytes": 3528,
|
| 25 |
+
"sha256": "a5dccd1fecd77d5c6fd4be4c27f148ff4df9e0597a5b12bba60fff329a8f6d84"
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp.pt",
|
| 29 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/ncp.pt",
|
| 30 |
+
"size_bytes": 427469662,
|
| 31 |
+
"sha256": "59e5405c493f4c5310dd9f9c53e3a5563de8902b387c281bb079dfa347a5c88d"
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae-config.yaml",
|
| 35 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae-config.yaml",
|
| 36 |
+
"size_bytes": 2204,
|
| 37 |
+
"sha256": "42de51ae35c5eabaadf092d006402364b3d9abcf5c88e3729e5bb504347fe2a8"
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae.pt",
|
| 41 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-44000/vqvae.pt",
|
| 42 |
+
"size_bytes": 1300864085,
|
| 43 |
+
"sha256": "f85bbb9c51879b97f5dfa175815335643d2c18169568455e88731b84715fcdab"
|
| 44 |
+
},
|
| 45 |
+
{
|
| 46 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/manual-checkpoints/checkpoint-iter-45000.pt",
|
| 47 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-45000.pt",
|
| 48 |
+
"size_bytes": 2565487243,
|
| 49 |
+
"sha256": "729d7432a0d99e7e1ccacd1526ffe0426e374df87c9037da52a0af0a60dee726"
|
| 50 |
+
},
|
| 51 |
+
{
|
| 52 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp-config.yaml",
|
| 53 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp-config.yaml",
|
| 54 |
+
"size_bytes": 3528,
|
| 55 |
+
"sha256": "15f635dc8642a4d907d7591ace6cbfbf8d4453aea3417cd849664d0be56ec0f8"
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp.pt",
|
| 59 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/ncp.pt",
|
| 60 |
+
"size_bytes": 427469662,
|
| 61 |
+
"sha256": "ed0cf1f72beda0fc582af91014fd1dc7dcc8a2a23c7f38b91c853020da1db04a"
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae-config.yaml",
|
| 65 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae-config.yaml",
|
| 66 |
+
"size_bytes": 2204,
|
| 67 |
+
"sha256": "01fe4a876d94ccbdd2db2e1eb1875c79a9ab8cd847e674bdb7cba973c03a4827"
|
| 68 |
+
},
|
| 69 |
+
{
|
| 70 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae.pt",
|
| 71 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/exports/step-45000/vqvae.pt",
|
| 72 |
+
"size_bytes": 1300864085,
|
| 73 |
+
"sha256": "a16ecf6a1b8fb925e65bbcc120fce5fee7f0705e882c25351a1c4a6f6be0ceba"
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/EVAL_DONE.json",
|
| 77 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/EVAL_DONE.json",
|
| 78 |
+
"size_bytes": 97,
|
| 79 |
+
"sha256": "8c09c167a7014823111474214e1dc7d55933435a3817c760ddb3f76585652652"
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.json",
|
| 83 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.json",
|
| 84 |
+
"size_bytes": 139915,
|
| 85 |
+
"sha256": "d55c6f8d7ac4939703de6dbe18a7ea18df01a78000f399affec0045bcec58135"
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.samples.json",
|
| 89 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-0.samples.json",
|
| 90 |
+
"size_bytes": 140297,
|
| 91 |
+
"sha256": "891672e4d85cf834dab622f001a4ea6e089f20e6e68074cdca91cae25cf4a517"
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.json",
|
| 95 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.json",
|
| 96 |
+
"size_bytes": 140716,
|
| 97 |
+
"sha256": "f09db410bace324d426e2fece48cc8d63c9fd5efa2d6a401de9b884e2f8ce21d"
|
| 98 |
+
},
|
| 99 |
+
{
|
| 100 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.samples.json",
|
| 101 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-1.samples.json",
|
| 102 |
+
"size_bytes": 141102,
|
| 103 |
+
"sha256": "70515a25f18fe739ad3ed422dad50dadaa6a3187beff59b60f436ce1753e0031"
|
| 104 |
+
},
|
| 105 |
+
{
|
| 106 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.json",
|
| 107 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.json",
|
| 108 |
+
"size_bytes": 142006,
|
| 109 |
+
"sha256": "8739454acd33efe78f2d588916571ef6cfe993341a8f48b5828b1dc1842ea28c"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.samples.json",
|
| 113 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-2.samples.json",
|
| 114 |
+
"size_bytes": 142388,
|
| 115 |
+
"sha256": "89cf46368cd424c80e8e61ae8c1bca0950a54ee0a72cce66f54f5747cdbaa5fe"
|
| 116 |
+
},
|
| 117 |
+
{
|
| 118 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.json",
|
| 119 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.json",
|
| 120 |
+
"size_bytes": 137427,
|
| 121 |
+
"sha256": "0ceb6857343c0f15ff92fbe8fce64d72dcd7c2471c9969996438e53d0ced6ba9"
|
| 122 |
+
},
|
| 123 |
+
{
|
| 124 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.samples.json",
|
| 125 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-3.samples.json",
|
| 126 |
+
"size_bytes": 137813,
|
| 127 |
+
"sha256": "c0ddf888841b4e19fc3799c97cf481037cc44e87e42f76c4375483d284778cb3"
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.json",
|
| 131 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.json",
|
| 132 |
+
"size_bytes": 143033,
|
| 133 |
+
"sha256": "d70cbec97e155b63900639db7b1c150085699e55e7b7dfee54ae4e4a1c3a214e"
|
| 134 |
+
},
|
| 135 |
+
{
|
| 136 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.samples.json",
|
| 137 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/random-seed-4.samples.json",
|
| 138 |
+
"size_bytes": 143416,
|
| 139 |
+
"sha256": "abcb84609659b85030b8da9d3529e7c8aa54234e6b245bfd6d86d06103fdfc86"
|
| 140 |
+
},
|
| 141 |
+
{
|
| 142 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/summary.json",
|
| 143 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-25750/summary.json",
|
| 144 |
+
"size_bytes": 859,
|
| 145 |
+
"sha256": "24c26188704f1be2574c31246171bef7e0ced59f30bddbc991f9264ef0cf9a83"
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/STOPPED.json",
|
| 149 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/STOPPED.json",
|
| 150 |
+
"size_bytes": 397,
|
| 151 |
+
"sha256": "a9e18ead0604b4b8acbc2f04f17f9100840caf5324f829e894aa91828c48f3e9"
|
| 152 |
+
},
|
| 153 |
+
{
|
| 154 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/STOP_REQUESTED.json",
|
| 155 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/STOP_REQUESTED.json",
|
| 156 |
+
"size_bytes": 1047,
|
| 157 |
+
"sha256": "a91a6b80b5e16074974391e42eeff8c74c38b43912a6ca88cbfcbef959d372d5"
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/manifest.json",
|
| 161 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/manifest.json",
|
| 162 |
+
"size_bytes": 35157,
|
| 163 |
+
"sha256": "aaf11243953c582e83ba62448365d4039807f62205185e5f2228cda6348dd58c"
|
| 164 |
+
},
|
| 165 |
+
{
|
| 166 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/git-provenance.json",
|
| 167 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/git-provenance.json",
|
| 168 |
+
"size_bytes": 59,
|
| 169 |
+
"sha256": "96f601db3a051fd136086ca5f54c87ce2a72449e528d826a3b882541953e3fa2"
|
| 170 |
+
},
|
| 171 |
+
{
|
| 172 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/config.yaml",
|
| 173 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/config.yaml",
|
| 174 |
+
"size_bytes": 4572,
|
| 175 |
+
"sha256": "b72db37de5c634702e488e7f582115543c1ed551a01e1a4cef078a6d9dff2cc8"
|
| 176 |
+
},
|
| 177 |
+
{
|
| 178 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/progress.json",
|
| 179 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/progress.json",
|
| 180 |
+
"size_bytes": 402,
|
| 181 |
+
"sha256": "aa62ccc08980f01e258661f18f3918f58280eff4370febbd1a54d2d9256b3f57"
|
| 182 |
+
},
|
| 183 |
+
{
|
| 184 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/hf-stop-backup-20260917/validation-curve.json",
|
| 185 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/validation-curve.json",
|
| 186 |
+
"size_bytes": 11963,
|
| 187 |
+
"sha256": "d7dc5f8e605ecebb236ff081e2cf86af46e22c6f83304cf4e6450852a7c283ae"
|
| 188 |
+
},
|
| 189 |
+
{
|
| 190 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/training-to-12875.log",
|
| 191 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-12875.log",
|
| 192 |
+
"size_bytes": 507886,
|
| 193 |
+
"sha256": "d7e0f5288e8cecf8d031255bc9afa7606ba49fd4cb2b2347d1f53753f1cbd572"
|
| 194 |
+
},
|
| 195 |
+
{
|
| 196 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/training-to-25750.log",
|
| 197 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-25750.log",
|
| 198 |
+
"size_bytes": 803561,
|
| 199 |
+
"sha256": "79b2b12a7b102153c5bf4102e4d1e89259c491e4b128abddcd4a24d5ec6d8351"
|
| 200 |
+
},
|
| 201 |
+
{
|
| 202 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/training-to-4769.log",
|
| 203 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-4769.log",
|
| 204 |
+
"size_bytes": 242468,
|
| 205 |
+
"sha256": "f48339ccc1150a063d5450a8c841b481a45c4719c1dcbf14aad03e785aac122b"
|
| 206 |
+
},
|
| 207 |
+
{
|
| 208 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/training-to-51499.log",
|
| 209 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-51499.log",
|
| 210 |
+
"size_bytes": 1217730,
|
| 211 |
+
"sha256": "23fc7b806f53de781d2f9a673229461c05f412f38aee8016f87459dedb6c0c93"
|
| 212 |
+
},
|
| 213 |
+
{
|
| 214 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/training-to-954.log",
|
| 215 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-954.log",
|
| 216 |
+
"size_bytes": 67236,
|
| 217 |
+
"sha256": "b2fd347d43dc10e8257954e96162cb24602072da8067cfbf80cc2b3a0ec5f3dd"
|
| 218 |
+
},
|
| 219 |
+
{
|
| 220 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/milestone-iter-12875.pt",
|
| 221 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/milestone-iter-12875.pt",
|
| 222 |
+
"size_bytes": 2565487435,
|
| 223 |
+
"sha256": "fefa9b7bf94073fe1f4145462c8fd497bf2f6a5dd9bfe611bada5b1b1658942f"
|
| 224 |
+
},
|
| 225 |
+
{
|
| 226 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-12875-fp32/summary.json",
|
| 227 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-12875-fp32/summary.json",
|
| 228 |
+
"size_bytes": 969,
|
| 229 |
+
"sha256": "f677d9e6853198e694f03243866c2e0ef6956fcff661b1d6fe61e5f57e285040"
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/evaluation/step-22000-fp32-fp64/comparison.json",
|
| 233 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/evaluation/step-22000-fp32-fp64/comparison.json",
|
| 234 |
+
"size_bytes": 2475,
|
| 235 |
+
"sha256": "84381543ba0ce4468ef888334d228487752fc6a0b36da631a3bc00d69c561012"
|
| 236 |
+
},
|
| 237 |
+
{
|
| 238 |
+
"local_path": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/hf-stop-backup-20260917/README.md",
|
| 239 |
+
"repo_path": "iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/README.md",
|
| 240 |
+
"size_bytes": 1447,
|
| 241 |
+
"sha256": "2b790471d8f85682208c1654e09237a57f7daeb398b6300ed7b3c1836c619f29"
|
| 242 |
+
}
|
| 243 |
+
]
|
| 244 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/config.yaml
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
task: mstok-next-concept
|
| 2 |
+
experiment: iclr-debug-two-stage-135b-20260916-generator
|
| 3 |
+
experiment_dir: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator
|
| 4 |
+
dataset:
|
| 5 |
+
train_source: /home/ubuntu/data/full_owt/train_gpt2.bin
|
| 6 |
+
validate_source: /home/ubuntu/data/full_owt/valid_gpt2.bin
|
| 7 |
+
pad_token_id: 50257
|
| 8 |
+
pre_tokenizer:
|
| 9 |
+
name: hf
|
| 10 |
+
tokenizer: hf
|
| 11 |
+
model_id: gpt2
|
| 12 |
+
special_tokens:
|
| 13 |
+
eos_token: <|endoftext|>
|
| 14 |
+
pad_token: <|pad|>
|
| 15 |
+
codec:
|
| 16 |
+
n_layers: 6
|
| 17 |
+
context_length: 256
|
| 18 |
+
embed_dim: 384
|
| 19 |
+
in_vocab_size: 50304
|
| 20 |
+
vqvae_vocab_size:
|
| 21 |
+
- 16384
|
| 22 |
+
- 16384
|
| 23 |
+
- 16384
|
| 24 |
+
- 16384
|
| 25 |
+
- 16384
|
| 26 |
+
- 16384
|
| 27 |
+
- 16384
|
| 28 |
+
- 16384
|
| 29 |
+
- 16384
|
| 30 |
+
- 16384
|
| 31 |
+
- 16384
|
| 32 |
+
- 16384
|
| 33 |
+
- 16384
|
| 34 |
+
- 16384
|
| 35 |
+
- 16384
|
| 36 |
+
- 16384
|
| 37 |
+
compression_factor: 1
|
| 38 |
+
dropout: 0.1
|
| 39 |
+
pre_quant_groupnorm: 4
|
| 40 |
+
pre_quant_dropout: 0.2
|
| 41 |
+
vector_quantizer_config:
|
| 42 |
+
clss: multiscale_residual_vector_quantizer
|
| 43 |
+
decay: 0.99
|
| 44 |
+
epsilon: 1.0e-05
|
| 45 |
+
commitment_cost: 0.25
|
| 46 |
+
learned_l1_sampling: true
|
| 47 |
+
learned_all_sampling: false
|
| 48 |
+
quant_resi:
|
| 49 |
+
enabled: true
|
| 50 |
+
ratio: 0.5
|
| 51 |
+
share_mode: 0
|
| 52 |
+
learnable_ratio: false
|
| 53 |
+
levels:
|
| 54 |
+
use_manual_levels: true
|
| 55 |
+
manual_levels:
|
| 56 |
+
- 1
|
| 57 |
+
- 4
|
| 58 |
+
- 9
|
| 59 |
+
- 16
|
| 60 |
+
- 25
|
| 61 |
+
- 36
|
| 62 |
+
- 49
|
| 63 |
+
- 64
|
| 64 |
+
- 81
|
| 65 |
+
- 100
|
| 66 |
+
- 121
|
| 67 |
+
- 144
|
| 68 |
+
- 169
|
| 69 |
+
- 196
|
| 70 |
+
- 225
|
| 71 |
+
- 256
|
| 72 |
+
aux:
|
| 73 |
+
fine_drop:
|
| 74 |
+
prob: 0.5
|
| 75 |
+
min_keep: 1
|
| 76 |
+
generator:
|
| 77 |
+
n_layer: 12
|
| 78 |
+
n_head: 12
|
| 79 |
+
bias: true
|
| 80 |
+
dropout: 0.1
|
| 81 |
+
n_embd: 768
|
| 82 |
+
context_length: 256
|
| 83 |
+
vocab_size:
|
| 84 |
+
- 16384
|
| 85 |
+
- 16384
|
| 86 |
+
- 16384
|
| 87 |
+
- 16384
|
| 88 |
+
- 16384
|
| 89 |
+
- 16384
|
| 90 |
+
- 16384
|
| 91 |
+
- 16384
|
| 92 |
+
- 16384
|
| 93 |
+
- 16384
|
| 94 |
+
- 16384
|
| 95 |
+
- 16384
|
| 96 |
+
- 16384
|
| 97 |
+
- 16384
|
| 98 |
+
- 16384
|
| 99 |
+
attn_pattern: block_diagonal
|
| 100 |
+
use_positional_encoding: true
|
| 101 |
+
use_level_encoding: true
|
| 102 |
+
prefix_len: 0
|
| 103 |
+
use_rope: true
|
| 104 |
+
rope_base: 10000
|
| 105 |
+
use_qk_norm: true
|
| 106 |
+
post_upsample_conv:
|
| 107 |
+
enabled: true
|
| 108 |
+
kernel_size: 3
|
| 109 |
+
use_relu2: true
|
| 110 |
+
use_flex_attention: true
|
| 111 |
+
shared_output_head: true
|
| 112 |
+
shared_head_per_level_bias: false
|
| 113 |
+
shared_head_per_level_scale: false
|
| 114 |
+
shared_head_adapter_rank: 0
|
| 115 |
+
objectives:
|
| 116 |
+
codec_weight: 1.0
|
| 117 |
+
ncp_weight: 1.0
|
| 118 |
+
soft_assignment_temperature: 1.0
|
| 119 |
+
prediction_temperature: 1.0
|
| 120 |
+
mstok_weight: 0.25
|
| 121 |
+
residual_weight: 1.0
|
| 122 |
+
reconstruction_weight: 1.0
|
| 123 |
+
teacher_ema_decay: 0.999
|
| 124 |
+
teacher_ema_warmup_steps: 1000
|
| 125 |
+
mstok_warmup_steps: 500
|
| 126 |
+
optimization:
|
| 127 |
+
codec_lr: 0.001
|
| 128 |
+
codec_min_lr: 0.0001
|
| 129 |
+
codec_warmup_iters: 0
|
| 130 |
+
codec_lr_decay_iters: 128747
|
| 131 |
+
generator_lr: 0.0005
|
| 132 |
+
generator_min_lr: 1.0e-05
|
| 133 |
+
generator_warmup_iters: 225
|
| 134 |
+
generator_lr_decay_iters: 128747
|
| 135 |
+
beta_1: 0.9
|
| 136 |
+
codec_beta_2: 0.99
|
| 137 |
+
generator_beta_2: 0.99
|
| 138 |
+
weight_decay: 0.1
|
| 139 |
+
max_grad_norm: 1.0
|
| 140 |
+
level_loss_alpha: 1.0
|
| 141 |
+
grad_accumulation_steps: 16
|
| 142 |
+
corruption:
|
| 143 |
+
mode: per_level
|
| 144 |
+
per_level_probs:
|
| 145 |
+
- 0.85
|
| 146 |
+
- 0.8321428571
|
| 147 |
+
- 0.8142857143
|
| 148 |
+
- 0.7964285714
|
| 149 |
+
- 0.7785714286
|
| 150 |
+
- 0.7607142857
|
| 151 |
+
- 0.7428571429
|
| 152 |
+
- 0.725
|
| 153 |
+
- 0.7071428571
|
| 154 |
+
- 0.6892857143
|
| 155 |
+
- 0.6714285714
|
| 156 |
+
- 0.6535714286
|
| 157 |
+
- 0.6357142857
|
| 158 |
+
- 0.6178571429
|
| 159 |
+
- 0.6
|
| 160 |
+
skip_level0: true
|
| 161 |
+
training:
|
| 162 |
+
log_interval: 10
|
| 163 |
+
eval_interval: 1000
|
| 164 |
+
checkpoint_interval: 1000
|
| 165 |
+
eval_batch_size: 4
|
| 166 |
+
val_iters: 25
|
| 167 |
+
keep_last: 3
|
| 168 |
+
seed: 55
|
| 169 |
+
codec_initialization_seed: 42
|
| 170 |
+
generator_initialization_seed: 55
|
| 171 |
+
total_iters: 128747
|
| 172 |
+
batch_size: 32
|
| 173 |
+
expected_world_size: 8
|
| 174 |
+
expected_global_batch_size: 4096
|
| 175 |
+
resume_checkpoint: null
|
| 176 |
+
milestone_steps:
|
| 177 |
+
- 954
|
| 178 |
+
- 4769
|
| 179 |
+
- 12875
|
| 180 |
+
- 25750
|
| 181 |
+
- 51499
|
| 182 |
+
- 77248
|
| 183 |
+
- 102997
|
| 184 |
+
- 128747
|
| 185 |
+
generation_eval_steps:
|
| 186 |
+
- 12875
|
| 187 |
+
- 25750
|
| 188 |
+
- 51499
|
| 189 |
+
- 77248
|
| 190 |
+
- 102997
|
| 191 |
+
- 128747
|
| 192 |
+
torch_compile:
|
| 193 |
+
enable: true
|
| 194 |
+
scope: loss
|
| 195 |
+
mode: default
|
| 196 |
+
dynamic: false
|
| 197 |
+
fullgraph: true
|
| 198 |
+
backend: inductor
|
| 199 |
+
mixed_precision:
|
| 200 |
+
enable: true
|
| 201 |
+
wandb:
|
| 202 |
+
entity: mstok
|
| 203 |
+
project: iclr-debug
|
| 204 |
+
group: full-owt-ctx256-16sq-135b-v1
|
| 205 |
+
enable: true
|
| 206 |
+
id: iclr-debug-two-stage-135b-20260916-generator
|
| 207 |
+
resume: allow
|
| 208 |
+
gradients_and_params:
|
| 209 |
+
enable: false
|
| 210 |
+
log: all
|
| 211 |
+
log_freq: 1000
|
| 212 |
+
export:
|
| 213 |
+
vqvae_config_template: /home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/config/repro-ctx256/vqvae.yaml
|
| 214 |
+
ncp_config_template: /home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/config/repro-ctx256/ncp-sharedhead.yaml
|
| 215 |
+
iclr_debug:
|
| 216 |
+
stage: generator
|
| 217 |
+
target_positions: 135000000000
|
| 218 |
+
lr_horizon_positions: 135000000000
|
| 219 |
+
stop_step: 128747
|
| 220 |
+
pilot: false
|
| 221 |
+
tokenizer_checkpoint: /home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/tokenizer/milestone-iter-128747.pt
|
| 222 |
+
tokenizer_steps: 128747
|
| 223 |
+
budget_unit: input positions including padding; log non-padding tokens separately
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/git-provenance.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"commit": "1b5226d4bbdd34653cb5fdd58d436355fb76ac8c"
|
| 3 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/manifest.json
ADDED
|
@@ -0,0 +1,756 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format": "iclr-debug-production-v1",
|
| 3 |
+
"variant": "two-stage",
|
| 4 |
+
"configs": {
|
| 5 |
+
"tokenizer": {
|
| 6 |
+
"task": "mstok-next-concept",
|
| 7 |
+
"experiment": "iclr-debug-two-stage-135b-20260916-tokenizer",
|
| 8 |
+
"experiment_dir": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/tokenizer",
|
| 9 |
+
"dataset": {
|
| 10 |
+
"train_source": "/home/ubuntu/data/full_owt/train_gpt2.bin",
|
| 11 |
+
"validate_source": "/home/ubuntu/data/full_owt/valid_gpt2.bin",
|
| 12 |
+
"pad_token_id": 50257
|
| 13 |
+
},
|
| 14 |
+
"pre_tokenizer": {
|
| 15 |
+
"name": "hf",
|
| 16 |
+
"tokenizer": "hf",
|
| 17 |
+
"model_id": "gpt2",
|
| 18 |
+
"special_tokens": {
|
| 19 |
+
"eos_token": "<|endoftext|>",
|
| 20 |
+
"pad_token": "<|pad|>"
|
| 21 |
+
}
|
| 22 |
+
},
|
| 23 |
+
"codec": {
|
| 24 |
+
"n_layers": 6,
|
| 25 |
+
"context_length": 256,
|
| 26 |
+
"embed_dim": 384,
|
| 27 |
+
"in_vocab_size": 50304,
|
| 28 |
+
"vqvae_vocab_size": [
|
| 29 |
+
16384,
|
| 30 |
+
16384,
|
| 31 |
+
16384,
|
| 32 |
+
16384,
|
| 33 |
+
16384,
|
| 34 |
+
16384,
|
| 35 |
+
16384,
|
| 36 |
+
16384,
|
| 37 |
+
16384,
|
| 38 |
+
16384,
|
| 39 |
+
16384,
|
| 40 |
+
16384,
|
| 41 |
+
16384,
|
| 42 |
+
16384,
|
| 43 |
+
16384,
|
| 44 |
+
16384
|
| 45 |
+
],
|
| 46 |
+
"compression_factor": 1,
|
| 47 |
+
"dropout": 0.1,
|
| 48 |
+
"pre_quant_groupnorm": 4,
|
| 49 |
+
"pre_quant_dropout": 0.2,
|
| 50 |
+
"vector_quantizer_config": {
|
| 51 |
+
"clss": "multiscale_residual_vector_quantizer",
|
| 52 |
+
"decay": 0.99,
|
| 53 |
+
"epsilon": 1e-05,
|
| 54 |
+
"commitment_cost": 0.25,
|
| 55 |
+
"learned_l1_sampling": true,
|
| 56 |
+
"learned_all_sampling": false,
|
| 57 |
+
"quant_resi": {
|
| 58 |
+
"enabled": true,
|
| 59 |
+
"ratio": 0.5,
|
| 60 |
+
"share_mode": 0,
|
| 61 |
+
"learnable_ratio": false
|
| 62 |
+
},
|
| 63 |
+
"levels": {
|
| 64 |
+
"use_manual_levels": true,
|
| 65 |
+
"manual_levels": [
|
| 66 |
+
1,
|
| 67 |
+
4,
|
| 68 |
+
9,
|
| 69 |
+
16,
|
| 70 |
+
25,
|
| 71 |
+
36,
|
| 72 |
+
49,
|
| 73 |
+
64,
|
| 74 |
+
81,
|
| 75 |
+
100,
|
| 76 |
+
121,
|
| 77 |
+
144,
|
| 78 |
+
169,
|
| 79 |
+
196,
|
| 80 |
+
225,
|
| 81 |
+
256
|
| 82 |
+
]
|
| 83 |
+
},
|
| 84 |
+
"aux": {
|
| 85 |
+
"fine_drop": {
|
| 86 |
+
"prob": 0.5,
|
| 87 |
+
"min_keep": 1
|
| 88 |
+
}
|
| 89 |
+
}
|
| 90 |
+
}
|
| 91 |
+
},
|
| 92 |
+
"generator": {
|
| 93 |
+
"n_layer": 12,
|
| 94 |
+
"n_head": 12,
|
| 95 |
+
"bias": true,
|
| 96 |
+
"dropout": 0.1,
|
| 97 |
+
"n_embd": 768,
|
| 98 |
+
"context_length": 256,
|
| 99 |
+
"vocab_size": [
|
| 100 |
+
16384,
|
| 101 |
+
16384,
|
| 102 |
+
16384,
|
| 103 |
+
16384,
|
| 104 |
+
16384,
|
| 105 |
+
16384,
|
| 106 |
+
16384,
|
| 107 |
+
16384,
|
| 108 |
+
16384,
|
| 109 |
+
16384,
|
| 110 |
+
16384,
|
| 111 |
+
16384,
|
| 112 |
+
16384,
|
| 113 |
+
16384,
|
| 114 |
+
16384
|
| 115 |
+
],
|
| 116 |
+
"attn_pattern": "block_diagonal",
|
| 117 |
+
"use_positional_encoding": true,
|
| 118 |
+
"use_level_encoding": true,
|
| 119 |
+
"prefix_len": 0,
|
| 120 |
+
"use_rope": true,
|
| 121 |
+
"rope_base": 10000,
|
| 122 |
+
"use_qk_norm": true,
|
| 123 |
+
"post_upsample_conv": {
|
| 124 |
+
"enabled": true,
|
| 125 |
+
"kernel_size": 3
|
| 126 |
+
},
|
| 127 |
+
"use_relu2": true,
|
| 128 |
+
"use_flex_attention": true,
|
| 129 |
+
"shared_output_head": true,
|
| 130 |
+
"shared_head_per_level_bias": false,
|
| 131 |
+
"shared_head_per_level_scale": false,
|
| 132 |
+
"shared_head_adapter_rank": 0
|
| 133 |
+
},
|
| 134 |
+
"objectives": {
|
| 135 |
+
"codec_weight": 1.0,
|
| 136 |
+
"ncp_weight": 1.0,
|
| 137 |
+
"soft_assignment_temperature": 1.0,
|
| 138 |
+
"prediction_temperature": 1.0,
|
| 139 |
+
"mstok_weight": 0.25,
|
| 140 |
+
"residual_weight": 1.0,
|
| 141 |
+
"reconstruction_weight": 1.0,
|
| 142 |
+
"teacher_ema_decay": 0.999,
|
| 143 |
+
"teacher_ema_warmup_steps": 1000,
|
| 144 |
+
"mstok_warmup_steps": 500
|
| 145 |
+
},
|
| 146 |
+
"optimization": {
|
| 147 |
+
"codec_lr": 0.001,
|
| 148 |
+
"codec_min_lr": 0.0001,
|
| 149 |
+
"codec_warmup_iters": 0,
|
| 150 |
+
"codec_lr_decay_iters": 128747,
|
| 151 |
+
"generator_lr": 0.0005,
|
| 152 |
+
"generator_min_lr": 1e-05,
|
| 153 |
+
"generator_warmup_iters": 225,
|
| 154 |
+
"generator_lr_decay_iters": 128747,
|
| 155 |
+
"beta_1": 0.9,
|
| 156 |
+
"codec_beta_2": 0.99,
|
| 157 |
+
"generator_beta_2": 0.99,
|
| 158 |
+
"weight_decay": 0.1,
|
| 159 |
+
"max_grad_norm": 1.0,
|
| 160 |
+
"level_loss_alpha": 1.0,
|
| 161 |
+
"grad_accumulation_steps": 1,
|
| 162 |
+
"corruption": {
|
| 163 |
+
"mode": "per_level",
|
| 164 |
+
"per_level_probs": [
|
| 165 |
+
0.85,
|
| 166 |
+
0.8321428571,
|
| 167 |
+
0.8142857143,
|
| 168 |
+
0.7964285714,
|
| 169 |
+
0.7785714286,
|
| 170 |
+
0.7607142857,
|
| 171 |
+
0.7428571429,
|
| 172 |
+
0.725,
|
| 173 |
+
0.7071428571,
|
| 174 |
+
0.6892857143,
|
| 175 |
+
0.6714285714,
|
| 176 |
+
0.6535714286,
|
| 177 |
+
0.6357142857,
|
| 178 |
+
0.6178571429,
|
| 179 |
+
0.6
|
| 180 |
+
],
|
| 181 |
+
"skip_level0": true
|
| 182 |
+
}
|
| 183 |
+
},
|
| 184 |
+
"training": {
|
| 185 |
+
"log_interval": 10,
|
| 186 |
+
"eval_interval": 1000,
|
| 187 |
+
"checkpoint_interval": 1000,
|
| 188 |
+
"eval_batch_size": 4,
|
| 189 |
+
"val_iters": 25,
|
| 190 |
+
"keep_last": 3,
|
| 191 |
+
"seed": 55,
|
| 192 |
+
"codec_initialization_seed": 42,
|
| 193 |
+
"generator_initialization_seed": 55,
|
| 194 |
+
"total_iters": 128747,
|
| 195 |
+
"batch_size": 512,
|
| 196 |
+
"expected_world_size": 8,
|
| 197 |
+
"expected_global_batch_size": 4096,
|
| 198 |
+
"resume_checkpoint": null,
|
| 199 |
+
"milestone_steps": [
|
| 200 |
+
954,
|
| 201 |
+
4769,
|
| 202 |
+
12875,
|
| 203 |
+
25750,
|
| 204 |
+
51499,
|
| 205 |
+
77248,
|
| 206 |
+
102997,
|
| 207 |
+
128747
|
| 208 |
+
],
|
| 209 |
+
"generation_eval_steps": []
|
| 210 |
+
},
|
| 211 |
+
"torch_compile": {
|
| 212 |
+
"enable": true,
|
| 213 |
+
"scope": "loss",
|
| 214 |
+
"mode": "default",
|
| 215 |
+
"dynamic": false,
|
| 216 |
+
"fullgraph": true,
|
| 217 |
+
"backend": "inductor"
|
| 218 |
+
},
|
| 219 |
+
"mixed_precision": {
|
| 220 |
+
"enable": true
|
| 221 |
+
},
|
| 222 |
+
"wandb": {
|
| 223 |
+
"entity": "mstok",
|
| 224 |
+
"project": "iclr-debug",
|
| 225 |
+
"group": "full-owt-ctx256-16sq-135b-v1",
|
| 226 |
+
"enable": true,
|
| 227 |
+
"id": "iclr-debug-two-stage-135b-20260916-tokenizer",
|
| 228 |
+
"resume": "allow",
|
| 229 |
+
"gradients_and_params": {
|
| 230 |
+
"enable": false,
|
| 231 |
+
"log": "all",
|
| 232 |
+
"log_freq": 1000
|
| 233 |
+
}
|
| 234 |
+
},
|
| 235 |
+
"export": {
|
| 236 |
+
"vqvae_config_template": "/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/config/repro-ctx256/vqvae.yaml",
|
| 237 |
+
"ncp_config_template": "/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/config/repro-ctx256/ncp-sharedhead.yaml"
|
| 238 |
+
},
|
| 239 |
+
"iclr_debug": {
|
| 240 |
+
"stage": "tokenizer",
|
| 241 |
+
"target_positions": 135000000000,
|
| 242 |
+
"lr_horizon_positions": 135000000000,
|
| 243 |
+
"stop_step": 128747,
|
| 244 |
+
"pilot": false,
|
| 245 |
+
"tokenizer_checkpoint": null,
|
| 246 |
+
"tokenizer_steps": null,
|
| 247 |
+
"budget_unit": "input positions including padding; log non-padding tokens separately"
|
| 248 |
+
}
|
| 249 |
+
},
|
| 250 |
+
"generator": {
|
| 251 |
+
"task": "mstok-next-concept",
|
| 252 |
+
"experiment": "iclr-debug-two-stage-135b-20260916-generator",
|
| 253 |
+
"experiment_dir": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator",
|
| 254 |
+
"dataset": {
|
| 255 |
+
"train_source": "/home/ubuntu/data/full_owt/train_gpt2.bin",
|
| 256 |
+
"validate_source": "/home/ubuntu/data/full_owt/valid_gpt2.bin",
|
| 257 |
+
"pad_token_id": 50257
|
| 258 |
+
},
|
| 259 |
+
"pre_tokenizer": {
|
| 260 |
+
"name": "hf",
|
| 261 |
+
"tokenizer": "hf",
|
| 262 |
+
"model_id": "gpt2",
|
| 263 |
+
"special_tokens": {
|
| 264 |
+
"eos_token": "<|endoftext|>",
|
| 265 |
+
"pad_token": "<|pad|>"
|
| 266 |
+
}
|
| 267 |
+
},
|
| 268 |
+
"codec": {
|
| 269 |
+
"n_layers": 6,
|
| 270 |
+
"context_length": 256,
|
| 271 |
+
"embed_dim": 384,
|
| 272 |
+
"in_vocab_size": 50304,
|
| 273 |
+
"vqvae_vocab_size": [
|
| 274 |
+
16384,
|
| 275 |
+
16384,
|
| 276 |
+
16384,
|
| 277 |
+
16384,
|
| 278 |
+
16384,
|
| 279 |
+
16384,
|
| 280 |
+
16384,
|
| 281 |
+
16384,
|
| 282 |
+
16384,
|
| 283 |
+
16384,
|
| 284 |
+
16384,
|
| 285 |
+
16384,
|
| 286 |
+
16384,
|
| 287 |
+
16384,
|
| 288 |
+
16384,
|
| 289 |
+
16384
|
| 290 |
+
],
|
| 291 |
+
"compression_factor": 1,
|
| 292 |
+
"dropout": 0.1,
|
| 293 |
+
"pre_quant_groupnorm": 4,
|
| 294 |
+
"pre_quant_dropout": 0.2,
|
| 295 |
+
"vector_quantizer_config": {
|
| 296 |
+
"clss": "multiscale_residual_vector_quantizer",
|
| 297 |
+
"decay": 0.99,
|
| 298 |
+
"epsilon": 1e-05,
|
| 299 |
+
"commitment_cost": 0.25,
|
| 300 |
+
"learned_l1_sampling": true,
|
| 301 |
+
"learned_all_sampling": false,
|
| 302 |
+
"quant_resi": {
|
| 303 |
+
"enabled": true,
|
| 304 |
+
"ratio": 0.5,
|
| 305 |
+
"share_mode": 0,
|
| 306 |
+
"learnable_ratio": false
|
| 307 |
+
},
|
| 308 |
+
"levels": {
|
| 309 |
+
"use_manual_levels": true,
|
| 310 |
+
"manual_levels": [
|
| 311 |
+
1,
|
| 312 |
+
4,
|
| 313 |
+
9,
|
| 314 |
+
16,
|
| 315 |
+
25,
|
| 316 |
+
36,
|
| 317 |
+
49,
|
| 318 |
+
64,
|
| 319 |
+
81,
|
| 320 |
+
100,
|
| 321 |
+
121,
|
| 322 |
+
144,
|
| 323 |
+
169,
|
| 324 |
+
196,
|
| 325 |
+
225,
|
| 326 |
+
256
|
| 327 |
+
]
|
| 328 |
+
},
|
| 329 |
+
"aux": {
|
| 330 |
+
"fine_drop": {
|
| 331 |
+
"prob": 0.5,
|
| 332 |
+
"min_keep": 1
|
| 333 |
+
}
|
| 334 |
+
}
|
| 335 |
+
}
|
| 336 |
+
},
|
| 337 |
+
"generator": {
|
| 338 |
+
"n_layer": 12,
|
| 339 |
+
"n_head": 12,
|
| 340 |
+
"bias": true,
|
| 341 |
+
"dropout": 0.1,
|
| 342 |
+
"n_embd": 768,
|
| 343 |
+
"context_length": 256,
|
| 344 |
+
"vocab_size": [
|
| 345 |
+
16384,
|
| 346 |
+
16384,
|
| 347 |
+
16384,
|
| 348 |
+
16384,
|
| 349 |
+
16384,
|
| 350 |
+
16384,
|
| 351 |
+
16384,
|
| 352 |
+
16384,
|
| 353 |
+
16384,
|
| 354 |
+
16384,
|
| 355 |
+
16384,
|
| 356 |
+
16384,
|
| 357 |
+
16384,
|
| 358 |
+
16384,
|
| 359 |
+
16384
|
| 360 |
+
],
|
| 361 |
+
"attn_pattern": "block_diagonal",
|
| 362 |
+
"use_positional_encoding": true,
|
| 363 |
+
"use_level_encoding": true,
|
| 364 |
+
"prefix_len": 0,
|
| 365 |
+
"use_rope": true,
|
| 366 |
+
"rope_base": 10000,
|
| 367 |
+
"use_qk_norm": true,
|
| 368 |
+
"post_upsample_conv": {
|
| 369 |
+
"enabled": true,
|
| 370 |
+
"kernel_size": 3
|
| 371 |
+
},
|
| 372 |
+
"use_relu2": true,
|
| 373 |
+
"use_flex_attention": true,
|
| 374 |
+
"shared_output_head": true,
|
| 375 |
+
"shared_head_per_level_bias": false,
|
| 376 |
+
"shared_head_per_level_scale": false,
|
| 377 |
+
"shared_head_adapter_rank": 0
|
| 378 |
+
},
|
| 379 |
+
"objectives": {
|
| 380 |
+
"codec_weight": 1.0,
|
| 381 |
+
"ncp_weight": 1.0,
|
| 382 |
+
"soft_assignment_temperature": 1.0,
|
| 383 |
+
"prediction_temperature": 1.0,
|
| 384 |
+
"mstok_weight": 0.25,
|
| 385 |
+
"residual_weight": 1.0,
|
| 386 |
+
"reconstruction_weight": 1.0,
|
| 387 |
+
"teacher_ema_decay": 0.999,
|
| 388 |
+
"teacher_ema_warmup_steps": 1000,
|
| 389 |
+
"mstok_warmup_steps": 500
|
| 390 |
+
},
|
| 391 |
+
"optimization": {
|
| 392 |
+
"codec_lr": 0.001,
|
| 393 |
+
"codec_min_lr": 0.0001,
|
| 394 |
+
"codec_warmup_iters": 0,
|
| 395 |
+
"codec_lr_decay_iters": 128747,
|
| 396 |
+
"generator_lr": 0.0005,
|
| 397 |
+
"generator_min_lr": 1e-05,
|
| 398 |
+
"generator_warmup_iters": 225,
|
| 399 |
+
"generator_lr_decay_iters": 128747,
|
| 400 |
+
"beta_1": 0.9,
|
| 401 |
+
"codec_beta_2": 0.99,
|
| 402 |
+
"generator_beta_2": 0.99,
|
| 403 |
+
"weight_decay": 0.1,
|
| 404 |
+
"max_grad_norm": 1.0,
|
| 405 |
+
"level_loss_alpha": 1.0,
|
| 406 |
+
"grad_accumulation_steps": 16,
|
| 407 |
+
"corruption": {
|
| 408 |
+
"mode": "per_level",
|
| 409 |
+
"per_level_probs": [
|
| 410 |
+
0.85,
|
| 411 |
+
0.8321428571,
|
| 412 |
+
0.8142857143,
|
| 413 |
+
0.7964285714,
|
| 414 |
+
0.7785714286,
|
| 415 |
+
0.7607142857,
|
| 416 |
+
0.7428571429,
|
| 417 |
+
0.725,
|
| 418 |
+
0.7071428571,
|
| 419 |
+
0.6892857143,
|
| 420 |
+
0.6714285714,
|
| 421 |
+
0.6535714286,
|
| 422 |
+
0.6357142857,
|
| 423 |
+
0.6178571429,
|
| 424 |
+
0.6
|
| 425 |
+
],
|
| 426 |
+
"skip_level0": true
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"training": {
|
| 430 |
+
"log_interval": 10,
|
| 431 |
+
"eval_interval": 1000,
|
| 432 |
+
"checkpoint_interval": 1000,
|
| 433 |
+
"eval_batch_size": 4,
|
| 434 |
+
"val_iters": 25,
|
| 435 |
+
"keep_last": 3,
|
| 436 |
+
"seed": 55,
|
| 437 |
+
"codec_initialization_seed": 42,
|
| 438 |
+
"generator_initialization_seed": 55,
|
| 439 |
+
"total_iters": 128747,
|
| 440 |
+
"batch_size": 32,
|
| 441 |
+
"expected_world_size": 8,
|
| 442 |
+
"expected_global_batch_size": 4096,
|
| 443 |
+
"resume_checkpoint": null,
|
| 444 |
+
"milestone_steps": [
|
| 445 |
+
954,
|
| 446 |
+
4769,
|
| 447 |
+
12875,
|
| 448 |
+
25750,
|
| 449 |
+
51499,
|
| 450 |
+
77248,
|
| 451 |
+
102997,
|
| 452 |
+
128747
|
| 453 |
+
],
|
| 454 |
+
"generation_eval_steps": [
|
| 455 |
+
12875,
|
| 456 |
+
25750,
|
| 457 |
+
51499,
|
| 458 |
+
77248,
|
| 459 |
+
102997,
|
| 460 |
+
128747
|
| 461 |
+
]
|
| 462 |
+
},
|
| 463 |
+
"torch_compile": {
|
| 464 |
+
"enable": true,
|
| 465 |
+
"scope": "loss",
|
| 466 |
+
"mode": "default",
|
| 467 |
+
"dynamic": false,
|
| 468 |
+
"fullgraph": true,
|
| 469 |
+
"backend": "inductor"
|
| 470 |
+
},
|
| 471 |
+
"mixed_precision": {
|
| 472 |
+
"enable": true
|
| 473 |
+
},
|
| 474 |
+
"wandb": {
|
| 475 |
+
"entity": "mstok",
|
| 476 |
+
"project": "iclr-debug",
|
| 477 |
+
"group": "full-owt-ctx256-16sq-135b-v1",
|
| 478 |
+
"enable": true,
|
| 479 |
+
"id": "iclr-debug-two-stage-135b-20260916-generator",
|
| 480 |
+
"resume": "allow",
|
| 481 |
+
"gradients_and_params": {
|
| 482 |
+
"enable": false,
|
| 483 |
+
"log": "all",
|
| 484 |
+
"log_freq": 1000
|
| 485 |
+
}
|
| 486 |
+
},
|
| 487 |
+
"export": {
|
| 488 |
+
"vqvae_config_template": "/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/config/repro-ctx256/vqvae.yaml",
|
| 489 |
+
"ncp_config_template": "/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/config/repro-ctx256/ncp-sharedhead.yaml"
|
| 490 |
+
},
|
| 491 |
+
"iclr_debug": {
|
| 492 |
+
"stage": "generator",
|
| 493 |
+
"target_positions": 135000000000,
|
| 494 |
+
"lr_horizon_positions": 135000000000,
|
| 495 |
+
"stop_step": 128747,
|
| 496 |
+
"pilot": false,
|
| 497 |
+
"tokenizer_checkpoint": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/tokenizer/milestone-iter-128747.pt",
|
| 498 |
+
"tokenizer_steps": 128747,
|
| 499 |
+
"budget_unit": "input positions including padding; log non-padding tokens separately"
|
| 500 |
+
}
|
| 501 |
+
}
|
| 502 |
+
},
|
| 503 |
+
"sources": {
|
| 504 |
+
"config/repro-ctx256/alignment-generator.yaml": "f62f36d078a3a373a19ea716a3ebc7eda5ae2fb7d670c5b935c734826bb76895",
|
| 505 |
+
"config/repro-ctx256/alignment-joint.yaml": "bb511950d20335e4237e406464ab26760a7fd9c9c9472c0507addd88d33d5178",
|
| 506 |
+
"config/repro-ctx256/alignment-tokenizer.yaml": "2d32955296f7d15990f284328c1cf8f52071abdba28033a65e272bb49cf5dd56",
|
| 507 |
+
"config/repro-ctx256/alignment.yaml": "3a486fa2ed28eca4fc406232669399b0625d274376ac43271f4ccf187b1e9077",
|
| 508 |
+
"config/repro-ctx256/joint.yaml": "8b4889111687a52e2d7d71995bda45a7f01f2777674c88191f338dca4c1732d2",
|
| 509 |
+
"config/repro-ctx256/mstok-semantic-eostok-matched10ep.yaml": "ae8a72c0c9f1cb5d2e301d83a1cdc08985a1f895bf627a5f75b1b3ce627f5e58",
|
| 510 |
+
"config/repro-ctx256/mstok-semantic-eostok.yaml": "2c9c48b21325c0e479990701b28116d78f611101724846747d173e52145fe15a",
|
| 511 |
+
"config/repro-ctx256/mstok-semantic-gear-matched10ep.yaml": "164cb8ac39fa407bb337f559ad8a9c62a34708fb900b8829d3ccf182961f2d0c",
|
| 512 |
+
"config/repro-ctx256/mstok-semantic-gear.yaml": "8cb8c998551daef0bf7fd9ec77b13b34c06662028d8820767e1432b5a9e8be54",
|
| 513 |
+
"config/repro-ctx256/mstok-semantic.yaml": "e0b00dbfa3eb838103f657c19bde20e36375c7fbe692623452ea093003f1ac98",
|
| 514 |
+
"config/repro-ctx256/mstok-w1-pilot.yaml": "21ee22eeccaaea2563c92ccc84f6df3871ea5da6a5ce980656088ea577dc1438",
|
| 515 |
+
"config/repro-ctx256/mstok-w1.yaml": "b6bf6079114b3c997fab89e85a53318bb2ec5dae55678513599d84152eab701b",
|
| 516 |
+
"config/repro-ctx256/mstok.yaml": "d9bb216f4eeb58857d72be399d5f8847d197ce70d0e504879f0788045483dc87",
|
| 517 |
+
"config/repro-ctx256/ncp-sharedhead.yaml": "86c9efbcb2b4f74b7253aa3bc4393de4a6a05e20ea30da573d1ca094ec600645",
|
| 518 |
+
"config/repro-ctx256/ncp.yaml": "93663654c4ce5ad0931db052edf507fdcb7c0ff9fc9722f4018820deee5c5acf",
|
| 519 |
+
"config/repro-ctx256/substitution-generator.yaml": "064f55887b9b2b8f3b0b52c094d4a193510dc1148ad7a5c7f53e317971f87c7c",
|
| 520 |
+
"config/repro-ctx256/substitution-joint.yaml": "b1576f4ecfe6fc6029a0cd868271b80a1d988b7f3667c6b1a8d24a6a4015df3b",
|
| 521 |
+
"config/repro-ctx256/substitution-tokenizer.yaml": "e29be5046f0fafbfe4c06973be4cd036a004e845348911cc0e6e05a7fca5d425",
|
| 522 |
+
"config/repro-ctx256/substitution.yaml": "ebba1c7e664de0e84913bdfd38e794837781c42c1b48cde90135ff9417856a95",
|
| 523 |
+
"config/repro-ctx256/vqvae.yaml": "05fde46e57c3e6b411e8a6afcac251f178e4b8a0856a0b99332bb284fb4710ce",
|
| 524 |
+
"data/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 525 |
+
"data/serialized_dataset.py": "5622c2d94236cd1b9b4ca18e8d5c272a86b54d193d99048fb9b7cd2687e8dbc6",
|
| 526 |
+
"data/tinysentences/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 527 |
+
"data/tinysentences/clean.py": "918a9a91aae484a5e06d6d212c2c0974946a4032e061c62a7b65851735bfc8dd",
|
| 528 |
+
"data/tinysentences/clean_parsed.py": "009a087f967e171bc6098484dbfd58635302900f6b07629629898f1d00fa70a3",
|
| 529 |
+
"data/tinysentences/generate_meta.py": "662714f1370e05567d412add168a7c3a05f4b1c9adb047fad374900bc1936a7e",
|
| 530 |
+
"data/tinysentences/generate_metadata_char.py": "afd3336ea03ff0653ccaf41c121d9b93debf7a380db6cd881b62b308ee6ee68c",
|
| 531 |
+
"data/tinysentences/generate_training_data_word.py": "82e2edcb7ea4db0609ccf2b047a0b207e7b8fcff4fc0b4ceee521ff278e8a8e8",
|
| 532 |
+
"data/tinysentences/parse_summaries.py": "2317159364dbbe7206f252d2939cab5d375395e0a536be0a1f60c5df88b31702",
|
| 533 |
+
"data/tinysentences/prepare.py": "35263b287f93337fda4207ed1991e9e8a820ab75139411bb7dd5f0a8a25bcbd9",
|
| 534 |
+
"data/tinysentences/sentences_only.py": "5e4d72bf612df27dad46cf606a40402606d897043ae24274dc2dfa88ec0b3e79",
|
| 535 |
+
"data/tinysentences/stats.py": "2665e9e6fac4a8f6c3e5bd8b07cbfd941ba6359a43981d89bdef58c136fdd3a9",
|
| 536 |
+
"data/tinysentences/summarize/merge_summaries_char.py": "bce94f4d9a824c2d25e45448023d59b7fbc42aba22b33efbc067cdbba93e1934",
|
| 537 |
+
"data/tinysentences/summarize/merge_summaries_word.py": "f6079110332cc38b2dd3571c30bedcca97098c39498f45eec1ef6e60cce7ac81",
|
| 538 |
+
"data/tinysentences/summarize/summarize_gemini.py": "282a438efce0dee351fd9833a59c56e4480bed4314dae40844201794ada3bd96",
|
| 539 |
+
"data/tinysentences/summarize/summarize_gpt.py": "fc08cbc37d6161e5b0057a1ede43272d96f7274524917d6c4d42c22f505c6e09",
|
| 540 |
+
"data/tinysentences/summarize/summarize_llama.py": "7f3f8230652a5b88ea2856b273d497cd5b633580b430c909e4f996cd0232832d",
|
| 541 |
+
"data/tinysentences/summarize/summarize_qwen.py": "b07e3c226ab0fdf5b21b025263adfa7c2625d640f0fac3c787bc4a2c9e32107f",
|
| 542 |
+
"data/tinysentences/summarize/summarize_together.py": "8c26ea052f2a64c2653d39f1ea5c13983dfbc9547f21cd7d7ea9d9b39033bffe",
|
| 543 |
+
"data/tinysentences/test.py": "9749244da77c078f550a78751664b45bb92067682f3aa27c2338419f1186c844",
|
| 544 |
+
"data/tinysentences/test_regex.py": "46652e189a0d5f5c0c7a2c90b857082004bf6727ef1d3ad42b631ac7bb0aa88f",
|
| 545 |
+
"data/tinysentences/tinysentences_char_dataset.py": "152ab9c30aec669a9db14df9e90412597870105eff9cfb737b876aad2323a206",
|
| 546 |
+
"data/tinysentences/tinysentences_dataset.py": "f90251d21871f6a7ad5ebd2ad0a6e59cd0b02ca245f7e46f93e6d970bf657bc7",
|
| 547 |
+
"data/tinysentences/tinysentences_vae.py": "30f7adde5413348faf5066448706747fe26a7fedd510ffea720aa0bf8283f6f8",
|
| 548 |
+
"data/tinysentences/to_csv.py": "c06c74648ed95b4494835c631c94fe1a00a9ff47a272f5e38546a7f94ff9be8b",
|
| 549 |
+
"data/tinysentences/tokenize/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 550 |
+
"data/tinysentences/tokenize/common.py": "e877c74df2a48d15991492b126cd0c5fe26731d61f3aa36d83ebf2cb9bda34a0",
|
| 551 |
+
"data/tinysentences/tokenize/tokenizer.py": "c16abc50b92a163f6978542b36b03c9a14aa00153c73117c4d52084d5de25c2a",
|
| 552 |
+
"data/tinysentences/tokenize/train_bpe.py": "fa8e55978dd5ca90117200648b468fb22822bf6981f15f0394c815e72eae1bdb",
|
| 553 |
+
"data/tinystories/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 554 |
+
"data/tinystories/download.py": "5afce37e912f6965b6722f0bc028076a655fea78304f1a213e4862dbe223cdcb",
|
| 555 |
+
"data/tinystories/stats.py": "216650fe19b78e8431c4512ff87a5c84af64d9e56ab8fdaf3952ca035725ea1b",
|
| 556 |
+
"data/wikitext/parse_wikitext.py": "ab583e0328f82afc3ba242410e8076be42ef1a16a51d66f3eb46a713cca591d0",
|
| 557 |
+
"evaluation/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 558 |
+
"evaluation/eval_random_gen_ppl_checkpoint.py": "221ea1917dc2d3bc19be39da0e0b8cca49295d0400a05da276cd50956876da8d",
|
| 559 |
+
"evaluation/gen_ppl.py": "e6853b14d27f93407505eb7607a1bd9fed49a7add1511ceca2b4a3076d533859",
|
| 560 |
+
"evaluation/gen_ppl_ncp.py": "96d68adee372e5186d0470659171eb03829e3d37f611e8f5917b45fb61e3a4ad",
|
| 561 |
+
"evaluation/gen_ppl_ntp.py": "b2638622bc041fa88e9d538be4229b3723ebec19c5b99fd6e5b73db4776dc952",
|
| 562 |
+
"evaluation/guess.py": "c27803225dd21b2deff1e7e494b9e3666180b93776340164ce6e953dacc50656",
|
| 563 |
+
"evaluation/ppl.py": "f3b581cb4849b0b2aa518729b8338838c7055697c3a26028dedb819c335ac838",
|
| 564 |
+
"evaluation/run_genppl_multiseed.py": "f8c9ac982f4caa4335b5f4d1a22d70f56dc11a706106a3dfb31d344eb8c93c2e",
|
| 565 |
+
"evaluation/run_genppl_release_ts.py": "d1bc5c00089118d9cf6c05c77459329ad40c2f8bc347b5aeedda8f5318de8796",
|
| 566 |
+
"evaluation/run_genppl_sweep.py": "ade4f39e373125848fdd32938e8d7bf86a0ec5108cca9bd75f0a966d092a798e",
|
| 567 |
+
"evaluation/show_8k_samples.py": "30ad825d5a24375e178750dfaaac45bc486bc3587f2e7d6d305d53bf37f8d226",
|
| 568 |
+
"evaluation/show_low_sample.py": "d696f1e54ca67e87565dce5864e0e6e2ad4194108cfb1c9c1e06eebd97008da6",
|
| 569 |
+
"evaluation/summarize_gen_ppl.py": "307f3d0d85884879295d5268aaf7f399f654b1b330aaa5728d6824ad24c6d15a",
|
| 570 |
+
"evaluation/sweep_gen_ppl_checkpoints.py": "3a9141e27d84b49c5493194dfc5a70bfa0a915cdf5ceb7a46afcf27cc6f33acb",
|
| 571 |
+
"evaluation/sweep_gen_ppl_ncp.py": "d91c2959cfed51d224ccb1c0d0f2c3b9a35c4027cc9ab8c65aa7603d587de761",
|
| 572 |
+
"evaluation/sweep_gen_ppl_ncp_scaling.py": "05249c1e00c5eb05042b6ea5b1ca008ecb8f0e4a77537ee755a27c153c25a17c",
|
| 573 |
+
"evaluation/sweep_gen_ppl_ncp_yolo.py": "4b26ef25b27af2f7fcbcb5ff3c312d47d2a4746ba759155d28802e978a681755",
|
| 574 |
+
"evaluation/sweep_gen_ppl_ntp.py": "14a04074741ee44580ec6557870d42534dd172d26c5be3e943ed5baf1b727303",
|
| 575 |
+
"evaluation/sweep_random_llama_yolo.py": "4062b54289fcd9407d15feb1ffd241b8722fd717c700350c22472fecec5ae46c",
|
| 576 |
+
"evaluation/val_oracle_ppl.py": "1a23f3c2897b6a6a3d239ed533979de35352d303bae094f4006bef4156f935c0",
|
| 577 |
+
"evaluation/val_vqvae_ppl.py": "4c293c502d3faf982d4ad66566b7a037faa7d28191d929de1621cb67f5ae8b5f",
|
| 578 |
+
"models/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 579 |
+
"models/common.py": "99aa0da356963a282936413d738ee1da72a3e3f277fecc28eddb553051582a4b",
|
| 580 |
+
"models/masked_lm.py": "80279614d74efc7b4c25306789d74e2f71045c54e4e4f485ee365f664705d7f6",
|
| 581 |
+
"models/next_concept.py": "743667aac3bdceb7bb7476a580700185b2638ddd9575e6357db57acc7d699339",
|
| 582 |
+
"models/next_token.py": "d0d292b9f5c032a351cbca1a8f0a6d41c3062890cb3e3f5b67d4e23a7cffeb77",
|
| 583 |
+
"models/quant.py": "b7b73abd4e55e10454cb680dea82124c38c222d37236c4c76b5c7d21123ac1b1",
|
| 584 |
+
"models/semantic_input.py": "b06658d4bc641397cbb973255eadfd2d6087fde244d801d3c70d29e3ce841bf2",
|
| 585 |
+
"models/vae.py": "02e2213a8878920374baf27dd10dfc03847dce866557c97e816c427d230c9ded",
|
| 586 |
+
"models/vqvae.py": "8eb3503eb1d990121b7d6138a4f8216aecfc72246d4a92bbecf98025c8a3d504",
|
| 587 |
+
"scripts/benchmark_iclr_debug.py": "dba86e8d578eec2188c633a479ccf673d58f3e5580b9074869cae1a352b8a043",
|
| 588 |
+
"scripts/benchmark_mstok_compile.py": "aad909da6f05ba25ac3d8f6b3cf450103bed55d0b93b1490764c0ef54f35b606",
|
| 589 |
+
"scripts/eval_owt_ckpts_gpt2_large.py": "4977a2d614698f60032fb6fd7fef614eb503d5b3883f499146a4fc048214a509",
|
| 590 |
+
"scripts/evaluate_mstok_sidecar.py": "9ba33e865fa0fad34bd2ab375d7fe38f37c441e2d96348f0e389e29dd9643455",
|
| 591 |
+
"scripts/evaluations/sweep_gen_ppl_shared_vocab_ncp.py": "2311ac71f9beba282426faa21390a5898c7d183a95f529755d6e1d6e8fc24050",
|
| 592 |
+
"scripts/neurips/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 593 |
+
"scripts/neurips/eval_gen_ppl_ncp_owt.py": "b92eff1ce198eb4ac8c88da7113b41d095c12acad92db541685175993db3535c",
|
| 594 |
+
"scripts/neurips/sweep_gen_ppl_owt_gpt2.py": "02adb1629cfa40ff51a75152e4da9c9ee00ead42b7ecda9e78cf7101e6826a67",
|
| 595 |
+
"scripts/preflight_owtsmall_mstok_ctx256.py": "a4f80921435b6cd136d089ec39bfa5442359d0dffbeefaaa1019178b8364657a",
|
| 596 |
+
"scripts/prepare_full_owt.py": "b1cb6b74397625c2b0643c4ee744ca3c89dfa6bbb7eeb7e9681e3a36f91a2c24",
|
| 597 |
+
"scripts/r1_lab/diagnose_at_40k.py": "b1582dcc29ccebc9255eaa77a7d56a7dbdef589ec5d5a9b0af150dede98874ab",
|
| 598 |
+
"scripts/r1_lab/inspect_summaries.py": "4154e8a6c1b3f68df5c623a57a1a7632b594c46c5f6d1d85cb08c476d6aa9cea",
|
| 599 |
+
"scripts/r1_lab/overnight_pipeline.py": "53f4e03ce1a5fb43fbca3ae455288f45ae8350b0a16f3d5318763011de6a8f96",
|
| 600 |
+
"scripts/r1_lab/prime_rope_qknorm_score_cache.py": "a791cdec03b415bfccdfd8d89c7e659aff8a2a5696d16ab7b4350ff608d0391b",
|
| 601 |
+
"scripts/r1_lab/sweep_gen_ppl_r1lab.py": "c7c0acf9aa6f07bb0a985cf0d5e5692c696e658ebb0470b7c92d9332565f9c06",
|
| 602 |
+
"scripts/r1_lab/sweep_gen_ppl_r1lab_v15.py": "524aa8680aa1b9fea2689509848c19e5a74a3ad72e78b0388bbca7acd43be22e",
|
| 603 |
+
"scripts/r1_lab/sweep_gen_ppl_rope_qknorm.py": "7822df057355b0a3a71ea5145742107c6e30f906859096237ff61d765fe36d50",
|
| 604 |
+
"scripts/r1_lab/train_variant.py": "9d0a2091784609ac9728226afd82f56fd4c1ac2dc08e89e1b17b60d34bb3d55e",
|
| 605 |
+
"scripts/run_alignment.py": "6312cd9d0976899763589fdcbd258c7681f6e9e96d7396cf35ff022d4695b862",
|
| 606 |
+
"scripts/run_iclr_debug.py": "a48a14067af036a0eba46d8de69922c615c05f6bbbab11587dca292ea57682ce",
|
| 607 |
+
"scripts/run_mstok_semantic.py": "8a7a38a50dd6694950bfa0e7b79718ec653c7e70be99fe13cbf2ee05c28f46a0",
|
| 608 |
+
"scripts/run_mstok_w1_pilot.py": "35224e462d0d11eb3e4326b0daa8bad7c99e1e30cdc6383310f8e2c8b5bb708f",
|
| 609 |
+
"scripts/run_substitution.py": "2cd22eaa58dab8b5231984eee340a3a2bdd0b88cc83333f7133447471e7f73f1",
|
| 610 |
+
"scripts/smoke_test_masked_l0.py": "d24f1270b3219a31212b68bcbb786f649c7c72f96e53f602d87ad6c17f3f73e5",
|
| 611 |
+
"scripts/test0_diagnostics.py": "ed389c666beef015ccac004f314a5a1428def303b46ac0e442f0236337344549",
|
| 612 |
+
"scripts/test0_r1_similarity.py": "4a486878bebd418a3b5454ddbf02585726ccbacc2700fdf591aa524766f9c9c9",
|
| 613 |
+
"scripts/test0_r1_uniqueness.py": "3ee59d4397d5b268151dbb8ca779c7f9bf8461d129d97fcd1a517c48ec2a69a1",
|
| 614 |
+
"scripts/test0_residual_norms.py": "3c4f68e5960e84796dc0bca4fd4583bf3b598552654f9b565fd5fcabd420f9e1",
|
| 615 |
+
"scripts/test_gen_equiv.py": "ae0e9c513422ee58bdc050ed6218ee54fc555dfb15ea8968c9ffc62a7a7ea99a",
|
| 616 |
+
"scripts/training_budget.py": "bb94c19c6dfbdd542946b10476b8d6958b8a189abd141cf62288fdb5fa9a2d38",
|
| 617 |
+
"scripts/vqvae_utilization.py": "5f90327f1507a5042b2f77b119fe4fc0df4e2131e5bbc21069dffd1fadc059c5",
|
| 618 |
+
"scripts/xsum/codec_ceiling.py": "ddb751e9152d5c88966e588d89cdceee4c5d91c4139aaedf369f65a45ecf2276",
|
| 619 |
+
"scripts/xsum/eval_merge.py": "6e313d6ed25227d0418b7422d58e09ceda4fc6dfae7d0fd663417b306241135c",
|
| 620 |
+
"scripts/xsum/eval_q0.py": "82fe04dccce28d7fc9c6ec8b4c9c91404a49527f9437913f44f51a6853ba9abc",
|
| 621 |
+
"scripts/xsum/eval_shard.py": "07192f6ba6219da1e8123ceeb99748c86da85322859ed5fa25cebcc3816f7ea8",
|
| 622 |
+
"scripts/xsum/eval_shard_cfg.py": "95ac4a8e2e3a73a21fa4303c62df80c086f23bcfe5ab34eb07ae90b585052ecb",
|
| 623 |
+
"scripts/xsum/eval_test.py": "7f92d578c31893e788e5fa7d9eef3098ba5602b3642873ce2ce031342c8be868",
|
| 624 |
+
"scripts/xsum/eval_uncond_rep.py": "ba4740b29bc091d866e2dfd91881e9a7d5c46a1bfa993d13eb63c0467397ec28",
|
| 625 |
+
"scripts/xsum/make_warmstart.py": "78793d6ef11fbf658c2837ef12cb843fc4817add8ba73c822b5e408901de3318",
|
| 626 |
+
"scripts/xsum/mbr_fast.py": "3db6e21d2dcce2e3bcf6f22474535d7eba0232b071b4271d2b172f787aed5c85",
|
| 627 |
+
"scripts/xsum/mbr_select.py": "107a061014c62395b7f5fd986fba9e55dd738054e2474022005cb7e9e0c9b8fb",
|
| 628 |
+
"scripts/xsum/mbr_shard_cfg.py": "3e7162375028c391276149eff288e9de40c874a3108ac53de0248198f4d08791",
|
| 629 |
+
"scripts/xsum/pool_append.py": "e50252f48c4d8d905be0de3eb56ca1d720f8f8545c6fdcb58ed27f0360c4e2ad",
|
| 630 |
+
"scripts/xsum/prep_xsum_le48.py": "3688ed1abc49d18f75d34f8382ff68a5afbd753b36cefdc0652d0bffb6a5f347",
|
| 631 |
+
"scripts/xsum/prep_xsum_mix30.py": "4bd47e52637c4b6bca5fd4956c1787b506a7f51b8a5bb0b68f19f728c4d48afb",
|
| 632 |
+
"scripts/xsum/prep_xsum_mix_split.py": "ec2fe74f08068ae8fda4ed59a86537a1d3f48276d205099bbfe16f8428ca36d5",
|
| 633 |
+
"scripts/xsum/prep_xsum_pass_uncond.py": "82ea50248d811e5182e63d0b4dfd6f995dd2f47eb963b5bb1c5d09bc533db7c1",
|
| 634 |
+
"scripts/xsum/prep_xsum_test_all.py": "7389fbdf5612ba524801c5b04109f810b139152ca50101b60acbb9b1e767b256",
|
| 635 |
+
"scripts/xsum/score_checkpoints.py": "41b412c06cd92cbd4caa6b586da7ddc7e913607a47c1a3aac0ee0a36a20a36be",
|
| 636 |
+
"tokenization/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 637 |
+
"tokenization/bpe_tokenizer.py": "0c8af84868a6aa40f981768f8ab49d28e501d55cce444edb1ad315de53c3445a",
|
| 638 |
+
"tokenization/build_word_vocab.py": "778f9666f4e7c642dc471773ee81c92c26975db37de1ebfffb605cae211351b5",
|
| 639 |
+
"tokenization/byte_tokenizer.py": "eb567a1772ae69738884ed9b94330aa39387c49e54d76e4b9752e67674fad810",
|
| 640 |
+
"tokenization/compression_ratio.py": "ed07ab1cc38d6176362631ce655bc974dd3623b87693f51279d5047e6bc6890f",
|
| 641 |
+
"tokenization/hf_tokenizer.py": "2ca8fae097f9c640bcb52b4788c37de8e0d9bb2fa971e30a5abb39f90ff97d4c",
|
| 642 |
+
"tokenization/tokenize.py": "f8ab54338613b6bd84513fdda85be173fd32274c35d97ef39a5c17d584f8108f",
|
| 643 |
+
"tokenization/tokenize_doc_level.py": "a06c81cf088fb81d2317ed1c3a1417a0e48ae0ea04962824f13f5d3beddc8902",
|
| 644 |
+
"tokenization/tokenize_vqvae.py": "14730eec3da87f6528e5fbd1166d5da79e10ac1700f7cd0be73bc40716bdebfc",
|
| 645 |
+
"tokenization/tokenizer.py": "e527d47e4ed8634c673d48a872a07c9cc8828d30988331de2d4bd3b7c21ddcb7",
|
| 646 |
+
"tokenization/train_bpe.py": "bf4e2a29eeba874024ed95f32fe720ae2320f72cb978361692db73e71be98683",
|
| 647 |
+
"tokenization/vqvae_tokenizer.py": "d7fa9d595865aa3e0b0d185bfd04721eb297fe5f657c4fca4ce69fc96aca308f",
|
| 648 |
+
"tokenization/word_tokenizer.py": "a405621bf35ed6cbb29abadd9d1d651e0783e6801fc7645b35e799d5de381ac2",
|
| 649 |
+
"train_mstok.py": "077b7c99c12daad2bbd986918d0f487411779de05a472af71124098da399b79d",
|
| 650 |
+
"trainer/alignment_trainer.py": "54f8ee7cca4d407223f2ce5146e6fc561ad748e20f2c702beb12ae040cd290c4",
|
| 651 |
+
"trainer/corruption.py": "f0f24c94652e6e9cf8e4ef30ae1a302e3232d64e765c931696d454db96e8b57d",
|
| 652 |
+
"trainer/joint_trainer.py": "8c8ebb55d21b6e223be1f7595c51f1b9257ccac40f8cff28260c1088656e2e49",
|
| 653 |
+
"trainer/mstok_trainer.py": "90712e396472d5dac9679b62af4f0b62afa6390a839d4d9bd9dae57406fe76a4",
|
| 654 |
+
"trainer/ncp_trainer.py": "73fc23f78f403dc39800b924dfb18ad42ffeab847fed30e931183c3d3acaee68",
|
| 655 |
+
"trainer/ntp_trainer.py": "5076d7aa29fac70869e1e309a15af6c4d3a91d6de8e00ac113a586841a9866e1",
|
| 656 |
+
"trainer/onpolicy.py": "6ff1fe4103d71001afe2d73e574d20c7242381e420929db14020141464ec6821",
|
| 657 |
+
"trainer/plain_two_stage.py": "4c5973ed87100b1b478aa1df0bb16cd3d1eb5844dffe5894003dd67c463f6c9a",
|
| 658 |
+
"trainer/scheduled_sampling.py": "f53515e23c2cec1231265f21a341ff77d11fa8a6129e0e382d540cc85571f5ed",
|
| 659 |
+
"trainer/semantic_mstok_trainer.py": "ebc02e80d279a445c99d8f09505cdfaf7632f1a661d6880f09cc1bd0fa90ef30",
|
| 660 |
+
"trainer/semantic_teacher.py": "7983b5da86d25954ed8267eee03d144c421305967705e0a242f4f025794ceeae",
|
| 661 |
+
"trainer/substitution_trainer.py": "33c7833b2e8dfabbb9ee56b258708055d3194b8d7453a1b8d93f59b44ff91465",
|
| 662 |
+
"utils/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
| 663 |
+
"utils/arg_utils.py": "e9eb7971f6273e9630b56fd87a03725b29183a8c80abb014c4c84f8b1bdb6445",
|
| 664 |
+
"utils/benchmark.py": "ccf436ef72a4e626c7f671d8e63a683057bf502a3852e5a7142410d1dc3e5501",
|
| 665 |
+
"utils/data.py": "78cce11e777813755cd171ccaa3247f89ab99ac4079e0157e55c72ce8a5a7fe5",
|
| 666 |
+
"utils/dist.py": "b2849d3ce03a8b1737e3dd61216e76e4a9e82d63ad1ea74cddf72e2285eaca64",
|
| 667 |
+
"utils/experiment_config.py": "616a70fefaac99ab0d856a1b6c36787a28fb9af8a4c55fc6d4765012fbaaeb38",
|
| 668 |
+
"utils/iclr_training.py": "10c6857fa2319952fe0df18067730b42e3ff755244e2b059f6c95882f8d27c8e",
|
| 669 |
+
"utils/logging.py": "901e67515cc65a35f575f05d16a3b9ef4a5b2cb0ea252cd7abbc80ecce9e88e8",
|
| 670 |
+
"utils/lr.py": "4d1ea2c25340afc4bdde8b582dab36f174e0b2965fa7bafa2a7539019dbe2b7e",
|
| 671 |
+
"utils/misc.py": "be4a323116a0956a77960d1baf613f268057d6a36023663d2af9f8e4728cbe77",
|
| 672 |
+
"utils/plot_levels.py": "916b39885ced191960c26c334a3b16f9c99751f6c29c46bd69d14fee414965f3",
|
| 673 |
+
"utils/registry.py": "f7a44c6d18433100873d5df462b9f08a8bdeb11cf3d92226334c7e5ece318318",
|
| 674 |
+
"utils/train.py": "43eef988000f3a5e20856653d725337a0fe0dbe276f0cf80e1b200292abe812d",
|
| 675 |
+
"train_iclr_debug.py": "e0a0adf4a30ef55925f9017de24d434299749332e58a507a6424265f7a5f796e",
|
| 676 |
+
"config/iclr-debug/README.md": "1c757d2ff37026d08c22324ec44cfa760b0599af62c8ad9b728ada471a8f2905",
|
| 677 |
+
"config/iclr-debug/benchmark-16sq.yaml": "18c9cef7032f9d3806d8b0c8b20fe0fc7d1ea98ad1d2440e24a67ee873c97f9b",
|
| 678 |
+
"config/iclr-debug/full-owt-audit.json": "fefa394834f6e9f65944d3b514633c29061071ebe6ca60fce21a6edef9748c1e",
|
| 679 |
+
"config/iclr-debug/training-defaults.yaml": "3b144b8d53f96fe035e36edca96aac456e6cf21202373a467a3cd875fe86b474",
|
| 680 |
+
"config/iclr-debug/NCM-BUDGET-AUDIT.md": "e734caeab7edb612de7ff508a8c0ec52445530ab7a35895b5b793d6a7b752e4b",
|
| 681 |
+
"config/iclr-debug/TRAINING.md": "91555457c5cf6af078075673ce96cb440bc8e22aeae5211c252d94f92ea87ac7",
|
| 682 |
+
"config/iclr-debug/BENCHMARK.md": "2781ea506212e12a0640a4b0d2a2c881355ec85756ca4014bccb1b1aedccdf59",
|
| 683 |
+
"config/iclr-debug/HF-MODEL-CARD.md": "de1fe176ebde262eacdf32af7d3f8c6497a764dd98e5a9094dad26de4aade38e"
|
| 684 |
+
},
|
| 685 |
+
"dataset": {
|
| 686 |
+
"dataset": "hazyresearch/ncm-tokenized-datasets",
|
| 687 |
+
"revision": "951e04517fd53d57521d8e09abd6511a662787d0",
|
| 688 |
+
"dtype": "little-endian uint16",
|
| 689 |
+
"eos_id": 50256,
|
| 690 |
+
"offset_units": "token positions; [0, each EOS position + 1] with final sentinel",
|
| 691 |
+
"sampling": "length-weighted within-segment windows; pad short segments with 50257",
|
| 692 |
+
"caveat": "EOS segments are not independently verified source-document identities",
|
| 693 |
+
"splits": {
|
| 694 |
+
"train": {
|
| 695 |
+
"bytes": 18070847356,
|
| 696 |
+
"tokens": 9035423678,
|
| 697 |
+
"sha256": "b24b78a07f56974fd0c9855dc16f25b836d2cb172be23eb5d3c650cd6d73f56a",
|
| 698 |
+
"min_token_id": 0,
|
| 699 |
+
"max_token_id": 50256,
|
| 700 |
+
"segments": 8009770,
|
| 701 |
+
"trailing_tokens": 0,
|
| 702 |
+
"eos_only_segments": 0,
|
| 703 |
+
"length_percentiles": {
|
| 704 |
+
"min": 132.0,
|
| 705 |
+
"p25": 416.0,
|
| 706 |
+
"p50": 718.0,
|
| 707 |
+
"p75": 1249.0,
|
| 708 |
+
"p99": 7610.0,
|
| 709 |
+
"max": 131288.0
|
| 710 |
+
},
|
| 711 |
+
"segments_shorter_than_256": 673011,
|
| 712 |
+
"expected_padding_fraction_at_256": 1.5955357567300495e-05,
|
| 713 |
+
"offsets_status": "created",
|
| 714 |
+
"offsets_sha256": "7e84564946b8cf8190dbdc1e86bdf2fe8cfd9235a517328c80b4df038ea0028b",
|
| 715 |
+
"offset_entries": 8009771
|
| 716 |
+
},
|
| 717 |
+
"valid": {
|
| 718 |
+
"bytes": 9186834,
|
| 719 |
+
"tokens": 4593417,
|
| 720 |
+
"sha256": "1f2ee17e08327a39d4e2417948d02c7b998d626d2dcb2661197a6e208f58266f",
|
| 721 |
+
"min_token_id": 0,
|
| 722 |
+
"max_token_id": 50256,
|
| 723 |
+
"segments": 3999,
|
| 724 |
+
"trailing_tokens": 0,
|
| 725 |
+
"eos_only_segments": 0,
|
| 726 |
+
"length_percentiles": {
|
| 727 |
+
"min": 143.0,
|
| 728 |
+
"p25": 433.0,
|
| 729 |
+
"p50": 734.0,
|
| 730 |
+
"p75": 1292.5,
|
| 731 |
+
"p99": 7122.719999999997,
|
| 732 |
+
"max": 24868.0
|
| 733 |
+
},
|
| 734 |
+
"segments_shorter_than_256": 363,
|
| 735 |
+
"expected_padding_fraction_at_256": 1.6064002171981108e-05,
|
| 736 |
+
"offsets_status": "created",
|
| 737 |
+
"offsets_sha256": "09a748ce1dcec7de36c77c48e39ccbf5c0293cc073689fa7c5f48bdfb9bb9445",
|
| 738 |
+
"offset_entries": 4000
|
| 739 |
+
}
|
| 740 |
+
},
|
| 741 |
+
"overlap": {
|
| 742 |
+
"exact_train_segment_matches": 0,
|
| 743 |
+
"unique_validation_segments_in_train": 0,
|
| 744 |
+
"unique_validation_segments": 3999,
|
| 745 |
+
"near_duplicates_checked": false
|
| 746 |
+
}
|
| 747 |
+
},
|
| 748 |
+
"hf_repo": null,
|
| 749 |
+
"generation_evaluation": true,
|
| 750 |
+
"versions": {
|
| 751 |
+
"torch": "2.13.0+cu130",
|
| 752 |
+
"transformers": "4.44.2",
|
| 753 |
+
"numpy": "1.26.4",
|
| 754 |
+
"hydra-core": "1.3.6"
|
| 755 |
+
}
|
| 756 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/progress.json
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 45000,
|
| 3 |
+
"checkpoint": "/home/ubuntu/mstok-results/iclr-debug-two-stage-135b-20260916/generator/checkpoint-iter-45000.pt",
|
| 4 |
+
"nonpadding_tokens": 47185162118,
|
| 5 |
+
"validation": {
|
| 6 |
+
"loss": 5.5970091938972475,
|
| 7 |
+
"reconstruction_loss": 0.0,
|
| 8 |
+
"vq_loss": 0.0,
|
| 9 |
+
"ncp_loss": 5.5970091938972475,
|
| 10 |
+
"mstok_loss": 0.0,
|
| 11 |
+
"semantic_loss": 0.0,
|
| 12 |
+
"ncp_accuracy": 34.62951505016722
|
| 13 |
+
}
|
| 14 |
+
}
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-12875.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-25750.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-4769.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-51499.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/training-to-954.log
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
wandb: Currently logged in as: iskhare (mstok). Use `wandb login --relogin` to force relogin
|
| 2 |
+
VQVAE, Total parameter count: 223.75M
|
| 3 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 4 |
+
VQVAE, Total parameter count: 223.75M
|
| 5 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 6 |
+
VQVAE, Total parameter count: 223.75M
|
| 7 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 8 |
+
VQVAE, Total parameter count: 223.75M
|
| 9 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 10 |
+
VQVAE, Total parameter count: 223.75M
|
| 11 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 12 |
+
VQVAE, Total parameter count: 223.75M
|
| 13 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 14 |
+
VQVAE, Total parameter count: 223.75M
|
| 15 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 16 |
+
wandb: Tracking run with wandb version 0.17.9
|
| 17 |
+
wandb: Run data is saved locally in /home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/wandb/run-20260916_151853-iclr-debug-two-stage-135b-20260916-generator
|
| 18 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 19 |
+
wandb: Syncing run iclr-debug-two-stage-135b-20260916-generator
|
| 20 |
+
wandb: ⭐️ View project at https://wandb.ai/mstok/iclr-debug
|
| 21 |
+
wandb: 🚀 View run at https://wandb.ai/mstok/iclr-debug/runs/iclr-debug-two-stage-135b-20260916-generator
|
| 22 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 23 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 24 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 25 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 26 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 27 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 28 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 29 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 30 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 31 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 32 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 33 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 34 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 35 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 36 |
+
VQVAE, Total parameter count: 223.75M
|
| 37 |
+
VQVAE, Non embedding parameter count: 204.34M
|
| 38 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 39 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 40 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 41 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 42 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 43 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 44 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 45 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 46 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 47 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 48 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 49 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 50 |
+
NCP Transformer, Total parameter count: 104.61M
|
| 51 |
+
NCP Transformer, Non embedding parameter count: 104.30M
|
| 52 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 53 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 54 |
+
/home/ubuntu/mstok-runs/iclr-debug-two-stage-20260916/train_iclr_debug.py:129: FutureWarning: `broadcast_buffers` is deprecated. Use `forward_sync_buffers` instead. IMPORTANT: unlike `broadcast_buffers=False`, `forward_sync_buffers=False` still syncs buffers at init. If you rely on buffers NOT being synced at init (e.g. rank-local buffers), keep using `broadcast_buffers=False` until `init_sync_buffers` is available.
|
| 55 |
+
ddp = DDP(model, device_ids=[local_rank] if device.type == "cuda" else None,
|
| 56 |
+
{"step": 1, "stop_at": 954, "target": 128747, "training/loss": 9.871083557605743, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.871083557605743, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.004180602006688963, "training/generator_lr": 4.444444444444444e-06, "training/grad_norm": 0.14711065590381622, "training/positions": 1048576, "training/nonpadding_tokens": 1048576, "throughput/steps_per_second": 0.04057870272065834, "throughput/raw_tokens_per_second": 42549.85378401704, "throughput/peak_reserved_gib": 23.4921875}
|
| 57 |
+
{"step": 10, "stop_at": 954, "target": 128747, "training/loss": 9.844660883148512, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.844660883148512, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.006423320791527313, "training/generator_lr": 2.4444444444444445e-05, "training/grad_norm": 0.11732996255159378, "training/positions": 10485760, "training/nonpadding_tokens": 10485517, "throughput/steps_per_second": 0.4079115964308802, "throughput/raw_tokens_per_second": 427715.296526003, "throughput/peak_reserved_gib": 24.275390625}
|
| 58 |
+
{"step": 20, "stop_at": 954, "target": 128747, "training/loss": 9.761483243852854, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.761483243852854, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.10068392035953178, "training/generator_lr": 4.666666666666667e-05, "training/grad_norm": 0.07661070674657822, "training/positions": 20971520, "training/nonpadding_tokens": 20971118, "throughput/steps_per_second": 0.40710573381980913, "throughput/raw_tokens_per_second": 426874.82896467246, "throughput/peak_reserved_gib": 24.275390625}
|
| 59 |
+
{"step": 30, "stop_at": 954, "target": 128747, "training/loss": 9.691915126144886, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.691915126144886, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.12911201400501673, "training/generator_lr": 6.88888888888889e-05, "training/grad_norm": 0.06473804265260696, "training/positions": 31457280, "training/nonpadding_tokens": 31456759, "throughput/steps_per_second": 0.4075730020312962, "throughput/raw_tokens_per_second": 427366.4180592442, "throughput/peak_reserved_gib": 24.275390625}
|
| 60 |
+
{"step": 40, "stop_at": 954, "target": 128747, "training/loss": 9.61763765886426, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.61763765886426, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.14390415969899664, "training/generator_lr": 9.111111111111112e-05, "training/grad_norm": 0.05417146906256676, "training/positions": 41943040, "training/nonpadding_tokens": 41942305, "throughput/steps_per_second": 0.4073528609142874, "throughput/raw_tokens_per_second": 427131.71613483626, "throughput/peak_reserved_gib": 24.275390625}
|
| 61 |
+
{"step": 50, "stop_at": 954, "target": 128747, "training/loss": 9.540263690799474, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.540263690799474, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.15787488242056857, "training/generator_lr": 0.00011333333333333333, "training/grad_norm": 0.046574149280786514, "training/positions": 52428800, "training/nonpadding_tokens": 52427940, "throughput/steps_per_second": 0.40740229747828155, "throughput/raw_tokens_per_second": 427187.17895186803, "throughput/peak_reserved_gib": 24.275390625}
|
| 62 |
+
{"step": 60, "stop_at": 954, "target": 128747, "training/loss": 9.472409730404616, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.472409730404616, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.2101503710284281, "training/generator_lr": 0.00013555555555555556, "training/grad_norm": 0.04066415876150131, "training/positions": 62914560, "training/nonpadding_tokens": 62913625, "throughput/steps_per_second": 0.40773109145099445, "throughput/raw_tokens_per_second": 427533.97896613204, "throughput/peak_reserved_gib": 24.275390625}
|
| 63 |
+
{"step": 70, "stop_at": 954, "target": 128747, "training/loss": 9.417974783480167, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.417974783480167, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.279105808423913, "training/generator_lr": 0.00015777777777777776, "training/grad_norm": 0.047600157558918, "training/positions": 73400320, "training/nonpadding_tokens": 73399129, "throughput/steps_per_second": 0.4075350686449746, "throughput/raw_tokens_per_second": 427321.0592417155, "throughput/peak_reserved_gib": 24.275390625}
|
| 64 |
+
{"step": 80, "stop_at": 954, "target": 128747, "training/loss": 9.377109003067016, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.377109003067016, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.29349459134615385, "training/generator_lr": 0.00017999999999999998, "training/grad_norm": 0.03833989426493645, "training/positions": 83886080, "training/nonpadding_tokens": 83884677, "throughput/steps_per_second": 0.4075239805912369, "throughput/raw_tokens_per_second": 427311.2259640483, "throughput/peak_reserved_gib": 24.275390625}
|
| 65 |
+
{"step": 90, "stop_at": 954, "target": 128747, "training/loss": 9.345919600129127, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.345919600129127, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.42629402696488294, "training/generator_lr": 0.00020222222222222223, "training/grad_norm": 0.05508098751306534, "training/positions": 94371840, "training/nonpadding_tokens": 94370271, "throughput/steps_per_second": 0.4076887400406052, "throughput/raw_tokens_per_second": 427485.86064373294, "throughput/peak_reserved_gib": 24.275390625}
|
| 66 |
+
{"step": 100, "stop_at": 954, "target": 128747, "training/loss": 9.32499888986349, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.32499888986349, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.560940439485786, "training/generator_lr": 0.00022444444444444446, "training/grad_norm": 0.053070779889822006, "training/positions": 104857600, "training/nonpadding_tokens": 104855825, "throughput/steps_per_second": 0.4077148678501129, "throughput/raw_tokens_per_second": 427511.6263445223, "throughput/peak_reserved_gib": 24.275390625}
|
| 67 |
+
{"step": 110, "stop_at": 954, "target": 128747, "training/loss": 9.301467409729957, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.301467409729957, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.7047286528010034, "training/generator_lr": 0.0002466666666666667, "training/grad_norm": 0.06223422288894653, "training/positions": 115343360, "training/nonpadding_tokens": 115341579, "throughput/steps_per_second": 0.40769303410734853, "throughput/raw_tokens_per_second": 427496.8863163266, "throughput/peak_reserved_gib": 24.275390625}
|
| 68 |
+
{"step": 120, "stop_at": 954, "target": 128747, "training/loss": 9.274017015099526, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.274017015099526, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 0.8985648777173914, "training/generator_lr": 0.00026888888888888893, "training/grad_norm": 0.13982895016670227, "training/positions": 125829120, "training/nonpadding_tokens": 125827327, "throughput/steps_per_second": 0.407567096817899, "throughput/raw_tokens_per_second": 427364.58703240904, "throughput/peak_reserved_gib": 24.275390625}
|
| 69 |
+
{"step": 130, "stop_at": 954, "target": 128747, "training/loss": 9.245860026031732, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.245860026031732, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 1.0634781955476589, "training/generator_lr": 0.00029111111111111113, "training/grad_norm": 0.13594666123390198, "training/positions": 136314880, "training/nonpadding_tokens": 136312889, "throughput/steps_per_second": 0.4075819043854277, "throughput/raw_tokens_per_second": 427372.53285114735, "throughput/peak_reserved_gib": 24.275390625}
|
| 70 |
+
{"step": 140, "stop_at": 954, "target": 128747, "training/loss": 9.211429408937693, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.211429408937693, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 1.2399420594690636, "training/generator_lr": 0.0003133333333333334, "training/grad_norm": 0.12958887219429016, "training/positions": 146800640, "training/nonpadding_tokens": 146798407, "throughput/steps_per_second": 0.4074633294391767, "throughput/raw_tokens_per_second": 427246.4075174417, "throughput/peak_reserved_gib": 24.275390625}
|
| 71 |
+
{"step": 150, "stop_at": 954, "target": 128747, "training/loss": 9.174244021624327, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.174244021624327, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 1.4370149848453178, "training/generator_lr": 0.0003355555555555556, "training/grad_norm": 0.15638326108455658, "training/positions": 157286400, "training/nonpadding_tokens": 157284143, "throughput/steps_per_second": 0.40757154239395027, "throughput/raw_tokens_per_second": 427368.759465577, "throughput/peak_reserved_gib": 24.275390625}
|
| 72 |
+
{"step": 160, "stop_at": 954, "target": 128747, "training/loss": 9.141117287427187, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.141117287427187, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 1.598191889632107, "training/generator_lr": 0.00035777777777777777, "training/grad_norm": 0.07898727059364319, "training/positions": 167772160, "training/nonpadding_tokens": 167769767, "throughput/steps_per_second": 0.40762492446874615, "throughput/raw_tokens_per_second": 427420.1691007672, "throughput/peak_reserved_gib": 24.275390625}
|
| 73 |
+
{"step": 170, "stop_at": 954, "target": 128747, "training/loss": 9.106735324859619, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.106735324859619, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 1.7872367527173914, "training/generator_lr": 0.00038, "training/grad_norm": 0.152617409825325, "training/positions": 178257920, "training/nonpadding_tokens": 178255314, "throughput/steps_per_second": 0.4078569739411609, "throughput/raw_tokens_per_second": 427660.3469537818, "throughput/peak_reserved_gib": 24.275390625}
|
| 74 |
+
{"step": 180, "stop_at": 954, "target": 128747, "training/loss": 9.07505406215787, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.07505406215787, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 1.9343106579222409, "training/generator_lr": 0.0004022222222222222, "training/grad_norm": 0.12514373660087585, "training/positions": 188743680, "training/nonpadding_tokens": 188740985, "throughput/steps_per_second": 0.4078358538169566, "throughput/raw_tokens_per_second": 427643.25851287006, "throughput/peak_reserved_gib": 24.275390625}
|
| 75 |
+
{"step": 190, "stop_at": 954, "target": 128747, "training/loss": 9.04273925423622, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.04273925423622, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.1132240933319397, "training/generator_lr": 0.00042444444444444447, "training/grad_norm": 0.08854150027036667, "training/positions": 199229440, "training/nonpadding_tokens": 199226665, "throughput/steps_per_second": 0.40812692470204626, "throughput/raw_tokens_per_second": 427948.8331809752, "throughput/peak_reserved_gib": 24.275390625}
|
| 76 |
+
{"step": 200, "stop_at": 954, "target": 128747, "training/loss": 9.015308622270823, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 9.015308622270823, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.225790068457358, "training/generator_lr": 0.00044666666666666666, "training/grad_norm": 0.12497750669717789, "training/positions": 209715200, "training/nonpadding_tokens": 209712037, "throughput/steps_per_second": 0.40810996751045797, "throughput/raw_tokens_per_second": 427918.4826255066, "throughput/peak_reserved_gib": 24.275390625}
|
| 77 |
+
{"step": 210, "stop_at": 954, "target": 128747, "training/loss": 8.979814836382866, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.979814836382866, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.3991348113503346, "training/generator_lr": 0.0004688888888888889, "training/grad_norm": 0.1000954806804657, "training/positions": 220200960, "training/nonpadding_tokens": 220197684, "throughput/steps_per_second": 0.40812199630957774, "throughput/raw_tokens_per_second": 427942.3186237535, "throughput/peak_reserved_gib": 24.275390625}
|
| 78 |
+
{"step": 220, "stop_at": 954, "target": 128747, "training/loss": 8.958572793006898, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.958572793006898, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.4984322742474916, "training/generator_lr": 0.0004911111111111111, "training/grad_norm": 0.1701861172914505, "training/positions": 230686720, "training/nonpadding_tokens": 230683308, "throughput/steps_per_second": 0.4082000847540737, "throughput/raw_tokens_per_second": 428023.260549935, "throughput/peak_reserved_gib": 24.275390625}
|
| 79 |
+
{"step": 230, "stop_at": 954, "target": 128747, "training/loss": 8.925016617774963, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.925016617774963, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.66686317673495, "training/generator_lr": 0.0004999999981701273, "training/grad_norm": 0.08466679602861404, "training/positions": 241172480, "training/nonpadding_tokens": 241168894, "throughput/steps_per_second": 0.4083869075949109, "throughput/raw_tokens_per_second": 428217.60408604913, "throughput/peak_reserved_gib": 24.275390625}
|
| 80 |
+
{"step": 240, "stop_at": 954, "target": 128747, "training/loss": 8.900285533070564, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.900285533070564, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.77888044784699, "training/generator_lr": 0.0004999999835311453, "training/grad_norm": 0.15479880571365356, "training/positions": 251658240, "training/nonpadding_tokens": 251654582, "throughput/steps_per_second": 0.40832028543840454, "throughput/raw_tokens_per_second": 428151.9117178053, "throughput/peak_reserved_gib": 24.275390625}
|
| 81 |
+
{"step": 250, "stop_at": 954, "target": 128747, "training/loss": 8.87479942664504, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.87479942664504, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 2.900166897470736, "training/generator_lr": 0.0004999999542531824, "training/grad_norm": 0.0811324343085289, "training/positions": 262144000, "training/nonpadding_tokens": 262140041, "throughput/steps_per_second": 0.4084259037338934, "throughput/raw_tokens_per_second": 428253.30681396864, "throughput/peak_reserved_gib": 24.275390625}
|
| 82 |
+
{"step": 260, "stop_at": 954, "target": 128747, "training/loss": 8.84714948758483, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.84714948758483, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.0518590614548495, "training/generator_lr": 0.0004999999103362402, "training/grad_norm": 0.13342252373695374, "training/positions": 272629760, "training/nonpadding_tokens": 272625566, "throughput/steps_per_second": 0.4083502604907626, "throughput/raw_tokens_per_second": 428176.68651324033, "throughput/peak_reserved_gib": 24.275390625}
|
| 83 |
+
{"step": 270, "stop_at": 954, "target": 128747, "training/loss": 8.82384713292122, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.82384713292122, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.1786456155936453, "training/generator_lr": 0.0004999998517803213, "training/grad_norm": 0.15548212826251984, "training/positions": 283115520, "training/nonpadding_tokens": 283111224, "throughput/steps_per_second": 0.4085605127784436, "throughput/raw_tokens_per_second": 428402.5809299389, "throughput/peak_reserved_gib": 24.275390625}
|
| 84 |
+
{"step": 280, "stop_at": 954, "target": 128747, "training/loss": 8.802917934954166, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.802917934954166, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.2786763168896322, "training/generator_lr": 0.0004999997785854293, "training/grad_norm": 0.09455807507038116, "training/positions": 293601280, "training/nonpadding_tokens": 293596820, "throughput/steps_per_second": 0.40845573756965664, "throughput/raw_tokens_per_second": 428290.18480374414, "throughput/peak_reserved_gib": 24.275390625}
|
| 85 |
+
{"step": 290, "stop_at": 954, "target": 128747, "training/loss": 8.785858216881753, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.785858216881753, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.3791038487667224, "training/generator_lr": 0.0004999996907515685, "training/grad_norm": 0.180866539478302, "training/positions": 304087040, "training/nonpadding_tokens": 304082455, "throughput/steps_per_second": 0.40844399653324864, "throughput/raw_tokens_per_second": 428279.46655889106, "throughput/peak_reserved_gib": 24.275390625}
|
| 86 |
+
{"step": 300, "stop_at": 954, "target": 128747, "training/loss": 8.776521760970354, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.776521760970354, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.4013377926421406, "training/generator_lr": 0.0004999995882787442, "training/grad_norm": 0.0964689552783966, "training/positions": 314572800, "training/nonpadding_tokens": 314568065, "throughput/steps_per_second": 0.40854529070440465, "throughput/raw_tokens_per_second": 428384.6585663012, "throughput/peak_reserved_gib": 24.275390625}
|
| 87 |
+
{"step": 310, "stop_at": 954, "target": 128747, "training/loss": 8.75076204687357, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.75076204687357, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.5759628448996654, "training/generator_lr": 0.0004999994711669624, "training/grad_norm": 0.11090482026338577, "training/positions": 325058560, "training/nonpadding_tokens": 325053596, "throughput/steps_per_second": 0.4087171378681868, "throughput/raw_tokens_per_second": 428561.62193481467, "throughput/peak_reserved_gib": 24.275390625}
|
| 88 |
+
{"step": 320, "stop_at": 954, "target": 128747, "training/loss": 8.754878564178943, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.754878564178943, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.524508779264214, "training/generator_lr": 0.0004999993394162303, "training/grad_norm": 0.12473094463348389, "training/positions": 335544320, "training/nonpadding_tokens": 335538862, "throughput/steps_per_second": 0.40904418284154637, "throughput/raw_tokens_per_second": 428893.70628462493, "throughput/peak_reserved_gib": 24.275390625}
|
| 89 |
+
{"step": 330, "stop_at": 954, "target": 128747, "training/loss": 8.726928801089525, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.726928801089525, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.699445743624582, "training/generator_lr": 0.0004999991930265556, "training/grad_norm": 0.0908852145075798, "training/positions": 346030080, "training/nonpadding_tokens": 346024551, "throughput/steps_per_second": 0.40904800709458533, "throughput/raw_tokens_per_second": 428915.01884636155, "throughput/peak_reserved_gib": 24.275390625}
|
| 90 |
+
{"step": 340, "stop_at": 954, "target": 128747, "training/loss": 8.706380737572909, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.706380737572909, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.82765141617893, "training/generator_lr": 0.000499999031997947, "training/grad_norm": 0.1375385820865631, "training/positions": 356515840, "training/nonpadding_tokens": 356510227, "throughput/steps_per_second": 0.4090061161624415, "throughput/raw_tokens_per_second": 428870.56160977244, "throughput/peak_reserved_gib": 24.275390625}
|
| 91 |
+
{"step": 350, "stop_at": 954, "target": 128747, "training/loss": 8.70576949045062, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.70576949045062, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.8153741638795986, "training/generator_lr": 0.0004999988563304143, "training/grad_norm": 0.12410085648298264, "training/positions": 367001600, "training/nonpadding_tokens": 366995799, "throughput/steps_per_second": 0.40923725987019566, "throughput/raw_tokens_per_second": 429108.6753451647, "throughput/peak_reserved_gib": 24.275390625}
|
| 92 |
+
{"step": 360, "stop_at": 954, "target": 128747, "training/loss": 8.678820344060659, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.678820344060659, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 3.9796571253135453, "training/generator_lr": 0.000499998666023968, "training/grad_norm": 0.09511373192071915, "training/positions": 377487360, "training/nonpadding_tokens": 377481238, "throughput/steps_per_second": 0.4091258551817619, "throughput/raw_tokens_per_second": 428986.4197831198, "throughput/peak_reserved_gib": 24.275390625}
|
| 93 |
+
{"step": 370, "stop_at": 954, "target": 128747, "training/loss": 8.670443738251924, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.670443738251924, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.045591424540134, "training/generator_lr": 0.0004999984610786194, "training/grad_norm": 0.18459655344486237, "training/positions": 387973120, "training/nonpadding_tokens": 387966997, "throughput/steps_per_second": 0.40907723861253203, "throughput/raw_tokens_per_second": 428948.5336476505, "throughput/peak_reserved_gib": 24.275390625}
|
| 94 |
+
{"step": 380, "stop_at": 954, "target": 128747, "training/loss": 8.654626427590847, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.654626427590847, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.126290107650502, "training/generator_lr": 0.0004999982414943807, "training/grad_norm": 0.09897475689649582, "training/positions": 398458880, "training/nonpadding_tokens": 398452694, "throughput/steps_per_second": 0.40907886517110287, "throughput/raw_tokens_per_second": 428947.70292880374, "throughput/peak_reserved_gib": 24.275390625}
|
| 95 |
+
{"step": 390, "stop_at": 954, "target": 128747, "training/loss": 8.644327249377966, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.644327249377966, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.198583494460703, "training/generator_lr": 0.000499998007271265, "training/grad_norm": 0.17077597975730896, "training/positions": 408944640, "training/nonpadding_tokens": 408938400, "throughput/steps_per_second": 0.40892889570047186, "throughput/raw_tokens_per_second": 428790.81752198125, "throughput/peak_reserved_gib": 24.275390625}
|
| 96 |
+
{"step": 400, "stop_at": 954, "target": 128747, "training/loss": 8.631650238484145, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.631650238484145, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.277098792851171, "training/generator_lr": 0.0004999977584092864, "training/grad_norm": 0.11580249667167664, "training/positions": 419430400, "training/nonpadding_tokens": 419424058, "throughput/steps_per_second": 0.40909368040483113, "throughput/raw_tokens_per_second": 428961.6422686361, "throughput/peak_reserved_gib": 24.275390625}
|
| 97 |
+
{"step": 410, "stop_at": 954, "target": 128747, "training/loss": 8.612663938105106, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.612663938105106, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.393990711224917, "training/generator_lr": 0.0004999974949084598, "training/grad_norm": 0.13520166277885437, "training/positions": 429916160, "training/nonpadding_tokens": 429909611, "throughput/steps_per_second": 0.40885654883759254, "throughput/raw_tokens_per_second": 428708.7012233665, "throughput/peak_reserved_gib": 24.275390625}
|
| 98 |
+
{"step": 420, "stop_at": 954, "target": 128747, "training/loss": 8.61954056546092, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.61954056546092, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.3708634902801, "training/generator_lr": 0.0004999972167688007, "training/grad_norm": 0.15599428117275238, "training/positions": 440401920, "training/nonpadding_tokens": 440395082, "throughput/steps_per_second": 0.40897724406762287, "throughput/raw_tokens_per_second": 428831.90323309816, "throughput/peak_reserved_gib": 24.275390625}
|
| 99 |
+
{"step": 430, "stop_at": 954, "target": 128747, "training/loss": 8.598354217410087, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.598354217410087, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.482114862040134, "training/generator_lr": 0.000499996923990326, "training/grad_norm": 0.15363338589668274, "training/positions": 450887680, "training/nonpadding_tokens": 450880746, "throughput/steps_per_second": 0.40917505604906723, "throughput/raw_tokens_per_second": 429047.21549116867, "throughput/peak_reserved_gib": 24.275390625}
|
| 100 |
+
{"step": 440, "stop_at": 954, "target": 128747, "training/loss": 8.581800411641598, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.581800411641598, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.585274482650502, "training/generator_lr": 0.0004999966165730531, "training/grad_norm": 0.15540584921836853, "training/positions": 461373440, "training/nonpadding_tokens": 461366347, "throughput/steps_per_second": 0.40911934925477783, "throughput/raw_tokens_per_second": 428986.2257665248, "throughput/peak_reserved_gib": 24.275390625}
|
| 101 |
+
{"step": 450, "stop_at": 954, "target": 128747, "training/loss": 8.573741032928229, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.573741032928229, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.626546496132943, "training/generator_lr": 0.0004999962945170003, "training/grad_norm": 0.18613947927951813, "training/positions": 471859200, "training/nonpadding_tokens": 471852012, "throughput/steps_per_second": 0.4091469820544181, "throughput/raw_tokens_per_second": 429017.81895836396, "throughput/peak_reserved_gib": 24.275390625}
|
| 102 |
+
{"step": 460, "stop_at": 954, "target": 128747, "training/loss": 8.566143275797367, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.566143275797367, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.687565321906354, "training/generator_lr": 0.0004999959578221868, "training/grad_norm": 0.2156212478876114, "training/positions": 482344960, "training/nonpadding_tokens": 482337515, "throughput/steps_per_second": 0.409071422447864, "throughput/raw_tokens_per_second": 428931.9627291345, "throughput/peak_reserved_gib": 24.275390625}
|
| 103 |
+
{"step": 470, "stop_at": 954, "target": 128747, "training/loss": 8.56047316044569, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.56047316044569, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.732621106814381, "training/generator_lr": 0.000499995606488633, "training/grad_norm": 0.14393538236618042, "training/positions": 492830720, "training/nonpadding_tokens": 492823060, "throughput/steps_per_second": 0.4091828948175741, "throughput/raw_tokens_per_second": 429050.565683994, "throughput/peak_reserved_gib": 24.275390625}
|
| 104 |
+
{"step": 480, "stop_at": 954, "target": 128747, "training/loss": 8.545621962845326, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.545621962845326, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.83139926055602, "training/generator_lr": 0.0004999952405163596, "training/grad_norm": 0.15548564493656158, "training/positions": 503316480, "training/nonpadding_tokens": 503308713, "throughput/steps_per_second": 0.4091654631167813, "throughput/raw_tokens_per_second": 429036.70658268675, "throughput/peak_reserved_gib": 24.275390625}
|
| 105 |
+
{"step": 490, "stop_at": 954, "target": 128747, "training/loss": 8.534178778529167, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.534178778529167, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.913188819502508, "training/generator_lr": 0.0004999948599053887, "training/grad_norm": 0.15385407209396362, "training/positions": 513802240, "training/nonpadding_tokens": 513794206, "throughput/steps_per_second": 0.4091965470146508, "throughput/raw_tokens_per_second": 429062.75293462916, "throughput/peak_reserved_gib": 24.275390625}
|
| 106 |
+
{"step": 500, "stop_at": 954, "target": 128747, "training/loss": 8.532354607433081, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.532354607433081, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.9142992919105355, "training/generator_lr": 0.0004999944646557427, "training/grad_norm": 0.17138919234275818, "training/positions": 524288000, "training/nonpadding_tokens": 524279876, "throughput/steps_per_second": 0.40913950559903667, "throughput/raw_tokens_per_second": 429010.18396746507, "throughput/peak_reserved_gib": 24.275390625}
|
| 107 |
+
{"step": 510, "stop_at": 954, "target": 128747, "training/loss": 8.519752291589976, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.519752291589976, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 4.992246289715719, "training/generator_lr": 0.0004999940547674456, "training/grad_norm": 0.23349744081497192, "training/positions": 534773760, "training/nonpadding_tokens": 534765323, "throughput/steps_per_second": 0.4092498402357639, "throughput/raw_tokens_per_second": 429116.7509550569, "throughput/peak_reserved_gib": 24.275390625}
|
| 108 |
+
{"step": 520, "stop_at": 954, "target": 128747, "training/loss": 8.511002310365438, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.511002310365438, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.077915969899665, "training/generator_lr": 0.0004999936302405217, "training/grad_norm": 0.16636767983436584, "training/positions": 545259520, "training/nonpadding_tokens": 545250857, "throughput/steps_per_second": 0.40923757937995164, "throughput/raw_tokens_per_second": 429107.45526661817, "throughput/peak_reserved_gib": 24.275390625}
|
| 109 |
+
{"step": 530, "stop_at": 954, "target": 128747, "training/loss": 8.50705091357231, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.50705091357231, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.103381715091973, "training/generator_lr": 0.0004999931910749963, "training/grad_norm": 0.22393527626991272, "training/positions": 555745280, "training/nonpadding_tokens": 555736516, "throughput/steps_per_second": 0.4091472577645206, "throughput/raw_tokens_per_second": 429017.8625703865, "throughput/peak_reserved_gib": 24.275390625}
|
| 110 |
+
{"step": 540, "stop_at": 954, "target": 128747, "training/loss": 8.50012828707695, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.50012828707695, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.162498040342809, "training/generator_lr": 0.0004999927372708958, "training/grad_norm": 0.16262026131153107, "training/positions": 566231040, "training/nonpadding_tokens": 566221693, "throughput/steps_per_second": 0.4092109445533431, "throughput/raw_tokens_per_second": 429064.91839789884, "throughput/peak_reserved_gib": 24.275390625}
|
| 111 |
+
{"step": 550, "stop_at": 954, "target": 128747, "training/loss": 8.485838788002729, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.485838788002729, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.25310115750418, "training/generator_lr": 0.0004999922688282472, "training/grad_norm": 0.22948366403579712, "training/positions": 576716800, "training/nonpadding_tokens": 576707372, "throughput/steps_per_second": 0.4091298984989042, "throughput/raw_tokens_per_second": 429000.4784962091, "throughput/peak_reserved_gib": 24.275390625}
|
| 112 |
+
{"step": 560, "stop_at": 954, "target": 128747, "training/loss": 8.479960573464632, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.479960573464632, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.297951178407191, "training/generator_lr": 0.0004999917857470785, "training/grad_norm": 0.15211758017539978, "training/positions": 587202560, "training/nonpadding_tokens": 587192883, "throughput/steps_per_second": 0.40900110950443674, "throughput/raw_tokens_per_second": 428858.56327209756, "throughput/peak_reserved_gib": 24.275390625}
|
| 113 |
+
{"step": 570, "stop_at": 954, "target": 128747, "training/loss": 8.469613495469094, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.469613495469094, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.360356461642977, "training/generator_lr": 0.0004999912880274186, "training/grad_norm": 0.16650056838989258, "training/positions": 597688320, "training/nonpadding_tokens": 597678457, "throughput/steps_per_second": 0.4090618506550601, "throughput/raw_tokens_per_second": 428924.83056205814, "throughput/peak_reserved_gib": 24.275390625}
|
| 114 |
+
{"step": 580, "stop_at": 954, "target": 128747, "training/loss": 8.464067248255015, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.464067248255015, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.423978365384615, "training/generator_lr": 0.0004999907756692973, "training/grad_norm": 0.1933707296848297, "training/positions": 608174080, "training/nonpadding_tokens": 608164023, "throughput/steps_per_second": 0.4092350995865209, "throughput/raw_tokens_per_second": 429106.16462310374, "throughput/peak_reserved_gib": 24.275390625}
|
| 115 |
+
{"step": 590, "stop_at": 954, "target": 128747, "training/loss": 8.47042583823204, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.47042583823204, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.397513195025083, "training/generator_lr": 0.0004999902486727452, "training/grad_norm": 0.19290679693222046, "training/positions": 618659840, "training/nonpadding_tokens": 618649723, "throughput/steps_per_second": 0.40923986685550484, "throughput/raw_tokens_per_second": 429116.64718867675, "throughput/peak_reserved_gib": 24.275390625}
|
| 116 |
+
{"step": 600, "stop_at": 954, "target": 128747, "training/loss": 8.451304657012225, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.451304657012225, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.502776181020067, "training/generator_lr": 0.0004999897070377936, "training/grad_norm": 0.14480124413967133, "training/positions": 629145600, "training/nonpadding_tokens": 629135483, "throughput/steps_per_second": 0.40929322195290685, "throughput/raw_tokens_per_second": 429175.04950249125, "throughput/peak_reserved_gib": 24.275390625}
|
| 117 |
+
{"step": 610, "stop_at": 954, "target": 128747, "training/loss": 8.44464293718338, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.44464293718338, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.5708073134406355, "training/generator_lr": 0.0004999891507644751, "training/grad_norm": 0.3166012763977051, "training/positions": 639631360, "training/nonpadding_tokens": 639621155, "throughput/steps_per_second": 0.4091030384379982, "throughput/raw_tokens_per_second": 428972.0275264242, "throughput/peak_reserved_gib": 24.275390625}
|
| 118 |
+
{"step": 620, "stop_at": 954, "target": 128747, "training/loss": 8.436087684333325, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.436087684333325, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.608579705790134, "training/generator_lr": 0.0004999885798528228, "training/grad_norm": 0.17485766112804413, "training/positions": 650117120, "training/nonpadding_tokens": 650106618, "throughput/steps_per_second": 0.4092073486197359, "throughput/raw_tokens_per_second": 429072.8513280342, "throughput/peak_reserved_gib": 24.275390625}
|
| 119 |
+
{"step": 630, "stop_at": 954, "target": 128747, "training/loss": 8.42601277679205, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.42601277679205, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.693419797763378, "training/generator_lr": 0.0004999879943028709, "training/grad_norm": 0.1830541342496872, "training/positions": 660602880, "training/nonpadding_tokens": 660592301, "throughput/steps_per_second": 0.409154719955319, "throughput/raw_tokens_per_second": 429026.66914052493, "throughput/peak_reserved_gib": 24.275390625}
|
| 120 |
+
{"step": 640, "stop_at": 954, "target": 128747, "training/loss": 8.422753217816354, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.422753217816354, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.717995531981606, "training/generator_lr": 0.0004999873941146543, "training/grad_norm": 0.2844255864620209, "training/positions": 671088640, "training/nonpadding_tokens": 671078026, "throughput/steps_per_second": 0.40924564228983107, "throughput/raw_tokens_per_second": 429123.7262499539, "throughput/peak_reserved_gib": 24.275390625}
|
| 121 |
+
{"step": 650, "stop_at": 954, "target": 128747, "training/loss": 8.41614634618163, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.41614634618163, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.770396765259197, "training/generator_lr": 0.0004999867792882088, "training/grad_norm": 0.1553221046924591, "training/positions": 681574400, "training/nonpadding_tokens": 681563648, "throughput/steps_per_second": 0.4093646113068772, "throughput/raw_tokens_per_second": 429244.257434084, "throughput/peak_reserved_gib": 24.275390625}
|
| 122 |
+
{"step": 660, "stop_at": 954, "target": 128747, "training/loss": 8.402774673700332, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.402774673700332, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.8563979541178925, "training/generator_lr": 0.0004999861498235714, "training/grad_norm": 0.15722794830799103, "training/positions": 692060160, "training/nonpadding_tokens": 692049393, "throughput/steps_per_second": 0.4092011433274176, "throughput/raw_tokens_per_second": 429077.88426397525, "throughput/peak_reserved_gib": 24.275390625}
|
| 123 |
+
{"step": 670, "stop_at": 954, "target": 128747, "training/loss": 8.388637253642083, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.388637253642083, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.953275240384615, "training/generator_lr": 0.0004999855057207795, "training/grad_norm": 0.30821430683135986, "training/positions": 702545920, "training/nonpadding_tokens": 702535094, "throughput/steps_per_second": 0.4091709109353163, "throughput/raw_tokens_per_second": 429044.38299653574, "throughput/peak_reserved_gib": 24.275390625}
|
| 124 |
+
{"step": 680, "stop_at": 954, "target": 128747, "training/loss": 8.392153864353896, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.392153864353896, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.929978182483278, "training/generator_lr": 0.0004999848469798716, "training/grad_norm": 0.2453027218580246, "training/positions": 713031680, "training/nonpadding_tokens": 713020784, "throughput/steps_per_second": 0.40933680746789036, "throughput/raw_tokens_per_second": 429217.88686979836, "throughput/peak_reserved_gib": 24.275390625}
|
| 125 |
+
{"step": 690, "stop_at": 954, "target": 128747, "training/loss": 8.389953873306514, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.389953873306514, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 5.976471049331104, "training/generator_lr": 0.0004999841736008871, "training/grad_norm": 0.18437980115413666, "training/positions": 723517440, "training/nonpadding_tokens": 723506441, "throughput/steps_per_second": 0.40930303971969956, "throughput/raw_tokens_per_second": 429181.1283558146, "throughput/peak_reserved_gib": 24.275390625}
|
| 126 |
+
{"step": 700, "stop_at": 954, "target": 128747, "training/loss": 8.373191752284765, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.373191752284765, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.08065295777592, "training/generator_lr": 0.0004999834855838663, "training/grad_norm": 0.22483088076114655, "training/positions": 734003200, "training/nonpadding_tokens": 733992123, "throughput/steps_per_second": 0.4092312249428552, "throughput/raw_tokens_per_second": 429106.8489221248, "throughput/peak_reserved_gib": 24.275390625}
|
| 127 |
+
{"step": 710, "stop_at": 954, "target": 128747, "training/loss": 8.372330252081156, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.372330252081156, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.109221493520067, "training/generator_lr": 0.0004999827829288502, "training/grad_norm": 0.30003583431243896, "training/positions": 744488960, "training/nonpadding_tokens": 744477778, "throughput/steps_per_second": 0.4093183601981357, "throughput/raw_tokens_per_second": 429197.1110203383, "throughput/peak_reserved_gib": 24.275390625}
|
| 128 |
+
{"step": 720, "stop_at": 954, "target": 128747, "training/loss": 8.363162723183631, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.363162723183631, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.140623366952341, "training/generator_lr": 0.0004999820656358808, "training/grad_norm": 0.2820779085159302, "training/positions": 754974720, "training/nonpadding_tokens": 754963460, "throughput/steps_per_second": 0.4090993166694366, "throughput/raw_tokens_per_second": 428968.53410130116, "throughput/peak_reserved_gib": 24.275390625}
|
| 129 |
+
{"step": 730, "stop_at": 954, "target": 128747, "training/loss": 8.363611906021834, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.363611906021834, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.17256414611204, "training/generator_lr": 0.000499981333705001, "training/grad_norm": 0.17488700151443481, "training/positions": 765460480, "training/nonpadding_tokens": 765449028, "throughput/steps_per_second": 0.4091599236355162, "throughput/raw_tokens_per_second": 429027.4202155012, "throughput/peak_reserved_gib": 24.275390625}
|
| 130 |
+
{"step": 740, "stop_at": 954, "target": 128747, "training/loss": 8.350775718688965, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.350775718688965, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.267040852320235, "training/generator_lr": 0.0004999805871362545, "training/grad_norm": 0.22609943151474, "training/positions": 775946240, "training/nonpadding_tokens": 775934485, "throughput/steps_per_second": 0.40918077226016875, "throughput/raw_tokens_per_second": 429044.7392760792, "throughput/peak_reserved_gib": 24.275390625}
|
| 131 |
+
{"step": 750, "stop_at": 954, "target": 128747, "training/loss": 8.339639697223902, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.339639697223902, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.3236831103678925, "training/generator_lr": 0.000499979825929686, "training/grad_norm": 0.17672383785247803, "training/positions": 786432000, "training/nonpadding_tokens": 786420144, "throughput/steps_per_second": 0.4093643800973939, "throughput/raw_tokens_per_second": 429245.52964476595, "throughput/peak_reserved_gib": 24.275390625}
|
| 132 |
+
{"step": 760, "stop_at": 954, "target": 128747, "training/loss": 8.341326329857111, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.341326329857111, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.354978835702341, "training/generator_lr": 0.0004999790500853408, "training/grad_norm": 0.274244099855423, "training/positions": 796917760, "training/nonpadding_tokens": 796905882, "throughput/steps_per_second": 0.40929869585648787, "throughput/raw_tokens_per_second": 429179.88884928176, "throughput/peak_reserved_gib": 24.275390625}
|
| 133 |
+
{"step": 770, "stop_at": 954, "target": 128747, "training/loss": 8.340317272394895, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.340317272394895, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.3556924775292645, "training/generator_lr": 0.0004999782596032654, "training/grad_norm": 0.1526513546705246, "training/positions": 807403520, "training/nonpadding_tokens": 807391508, "throughput/steps_per_second": 0.40932875458747153, "throughput/raw_tokens_per_second": 429206.8231650011, "throughput/peak_reserved_gib": 24.275390625}
|
| 134 |
+
{"step": 780, "stop_at": 954, "target": 128747, "training/loss": 8.323338445276022, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.323338445276022, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.458089464882943, "training/generator_lr": 0.000499977454483507, "training/grad_norm": 0.20543967187404633, "training/positions": 817889280, "training/nonpadding_tokens": 817877191, "throughput/steps_per_second": 0.40928246131017665, "throughput/raw_tokens_per_second": 429160.61467582773, "throughput/peak_reserved_gib": 24.275390625}
|
| 135 |
+
{"step": 790, "stop_at": 954, "target": 128747, "training/loss": 8.330054011195898, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.330054011195898, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.430244369251672, "training/generator_lr": 0.0004999766347261137, "training/grad_norm": 0.22034521400928497, "training/positions": 828375040, "training/nonpadding_tokens": 828362696, "throughput/steps_per_second": 0.40924190705703006, "throughput/raw_tokens_per_second": 429110.8062656024, "throughput/peak_reserved_gib": 24.275390625}
|
| 136 |
+
{"step": 800, "stop_at": 954, "target": 128747, "training/loss": 8.323832194507123, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.323832194507123, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.485776154891305, "training/generator_lr": 0.0004999758003311345, "training/grad_norm": 0.25817862153053284, "training/positions": 838860800, "training/nonpadding_tokens": 838848410, "throughput/steps_per_second": 0.40928149017738297, "throughput/raw_tokens_per_second": 429160.8651493847, "throughput/peak_reserved_gib": 24.275390625}
|
| 137 |
+
{"step": 810, "stop_at": 954, "target": 128747, "training/loss": 8.315256363153457, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.315256363153457, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.552758870714883, "training/generator_lr": 0.0004999749512986192, "training/grad_norm": 0.218985915184021, "training/positions": 849346560, "training/nonpadding_tokens": 849334036, "throughput/steps_per_second": 0.4093218547591466, "throughput/raw_tokens_per_second": 429199.5882630732, "throughput/peak_reserved_gib": 24.275390625}
|
| 138 |
+
{"step": 820, "stop_at": 954, "target": 128747, "training/loss": 8.306205031648279, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.306205031648279, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.62062016617893, "training/generator_lr": 0.0004999740876286187, "training/grad_norm": 0.26127687096595764, "training/positions": 859832320, "training/nonpadding_tokens": 859819502, "throughput/steps_per_second": 0.40917429395211635, "throughput/raw_tokens_per_second": 429038.31473089213, "throughput/peak_reserved_gib": 24.275390625}
|
| 139 |
+
{"step": 830, "stop_at": 954, "target": 128747, "training/loss": 8.306372378021479, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.306372378021479, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.6381901259406355, "training/generator_lr": 0.0004999732093211843, "training/grad_norm": 0.23748648166656494, "training/positions": 870318080, "training/nonpadding_tokens": 870305154, "throughput/steps_per_second": 0.4092527417241979, "throughput/raw_tokens_per_second": 429128.1829765819, "throughput/peak_reserved_gib": 24.275390625}
|
| 140 |
+
{"step": 840, "stop_at": 954, "target": 128747, "training/loss": 8.29961592927575, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.29961592927575, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.693166675376254, "training/generator_lr": 0.0004999723163763688, "training/grad_norm": 0.2031191736459732, "training/positions": 880803840, "training/nonpadding_tokens": 880790680, "throughput/steps_per_second": 0.40944199240633405, "throughput/raw_tokens_per_second": 429321.4656868418, "throughput/peak_reserved_gib": 24.275390625}
|
| 141 |
+
{"step": 850, "stop_at": 954, "target": 128747, "training/loss": 8.29398488998413, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.29398488998413, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.708626737562709, "training/generator_lr": 0.0004999714087942253, "training/grad_norm": 0.22629296779632568, "training/positions": 891289600, "training/nonpadding_tokens": 891276364, "throughput/steps_per_second": 0.40927898094010473, "throughput/raw_tokens_per_second": 429157.0061979961, "throughput/peak_reserved_gib": 24.275390625}
|
| 142 |
+
{"step": 860, "stop_at": 954, "target": 128747, "training/loss": 8.284341107308865, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.284341107308865, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.795767793687291, "training/generator_lr": 0.0004999704865748083, "training/grad_norm": 0.20574678480625153, "training/positions": 901775360, "training/nonpadding_tokens": 901761946, "throughput/steps_per_second": 0.4094457259278737, "throughput/raw_tokens_per_second": 429327.67337662453, "throughput/peak_reserved_gib": 24.275390625}
|
| 143 |
+
{"step": 870, "stop_at": 954, "target": 128747, "training/loss": 8.292009130865335, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.292009130865335, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.777637698578595, "training/generator_lr": 0.0004999695497181725, "training/grad_norm": 0.202379047870636, "training/positions": 912261120, "training/nonpadding_tokens": 912247625, "throughput/steps_per_second": 0.409328868169343, "throughput/raw_tokens_per_second": 429209.1117057048, "throughput/peak_reserved_gib": 24.275390625}
|
| 144 |
+
{"step": 880, "stop_at": 954, "target": 128747, "training/loss": 8.282498709857464, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.282498709857464, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.840077275815218, "training/generator_lr": 0.0004999685982243744, "training/grad_norm": 0.3429924249649048, "training/positions": 922746880, "training/nonpadding_tokens": 922733261, "throughput/steps_per_second": 0.40943453302632893, "throughput/raw_tokens_per_second": 429318.1479144064, "throughput/peak_reserved_gib": 24.275390625}
|
| 145 |
+
{"step": 890, "stop_at": 954, "target": 128747, "training/loss": 8.287351156026125, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.287351156026125, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.816794915342809, "training/generator_lr": 0.0004999676320934704, "training/grad_norm": 0.1516335904598236, "training/positions": 933232640, "training/nonpadding_tokens": 933218837, "throughput/steps_per_second": 0.4094868317085001, "throughput/raw_tokens_per_second": 429370.5294878688, "throughput/peak_reserved_gib": 24.275390625}
|
| 146 |
+
{"step": 900, "stop_at": 954, "target": 128747, "training/loss": 8.270814752578735, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.270814752578735, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.926107859531773, "training/generator_lr": 0.0004999666513255184, "training/grad_norm": 0.21166975796222687, "training/positions": 943718400, "training/nonpadding_tokens": 943704386, "throughput/steps_per_second": 0.4092797105615343, "throughput/raw_tokens_per_second": 429152.2459798785, "throughput/peak_reserved_gib": 24.275390625}
|
| 147 |
+
{"step": 910, "stop_at": 954, "target": 128747, "training/loss": 8.266571156680584, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.266571156680584, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 6.94475236465301, "training/generator_lr": 0.000499965655920577, "training/grad_norm": 0.2676384150981903, "training/positions": 954204160, "training/nonpadding_tokens": 954190088, "throughput/steps_per_second": 0.40948542781703595, "throughput/raw_tokens_per_second": 429374.216943195, "throughput/peak_reserved_gib": 24.275390625}
|
| 148 |
+
{"step": 920, "stop_at": 954, "target": 128747, "training/loss": 8.263330242037773, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.263330242037773, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 7.004203464673913, "training/generator_lr": 0.0004999646458787058, "training/grad_norm": 0.21519799530506134, "training/positions": 964689920, "training/nonpadding_tokens": 964675810, "throughput/steps_per_second": 0.409409437332112, "throughput/raw_tokens_per_second": 429295.3544040948, "throughput/peak_reserved_gib": 24.275390625}
|
| 149 |
+
{"step": 930, "stop_at": 954, "target": 128747, "training/loss": 8.25900956839323, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.25900956839323, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 7.038980847617057, "training/generator_lr": 0.0004999636211999648, "training/grad_norm": 0.17417152225971222, "training/positions": 975175680, "training/nonpadding_tokens": 975161320, "throughput/steps_per_second": 0.4093959066354175, "throughput/raw_tokens_per_second": 429272.48729847366, "throughput/peak_reserved_gib": 24.275390625}
|
| 150 |
+
{"step": 940, "stop_at": 954, "target": 128747, "training/loss": 8.252974602580071, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.252974602580071, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 7.0835581495610365, "training/generator_lr": 0.0004999625818844157, "training/grad_norm": 0.17934155464172363, "training/positions": 985661440, "training/nonpadding_tokens": 985647032, "throughput/steps_per_second": 0.4092302643027272, "throughput/raw_tokens_per_second": 429107.0693162278, "throughput/peak_reserved_gib": 24.275390625}
|
| 151 |
+
{"step": 950, "stop_at": 954, "target": 128747, "training/loss": 8.25698967576027, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.25698967576027, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 7.081216359218227, "training/generator_lr": 0.0004999615279321202, "training/grad_norm": 0.30399030447006226, "training/positions": 996147200, "training/nonpadding_tokens": 996132624, "throughput/steps_per_second": 0.40931891305256246, "throughput/raw_tokens_per_second": 429195.11201526446, "throughput/peak_reserved_gib": 24.275390625}
|
| 152 |
+
{"step": 954, "stop_at": 954, "target": 128747, "training/loss": 8.263838732615113, "training/reconstruction_loss": 0.0, "training/vq_loss": 0.0, "training/ncp_loss": 8.263838732615113, "training/mstok_loss": 0.0, "training/semantic_loss": 0.0, "training/ncp_accuracy": 7.037092227999582, "training/generator_lr": 0.0004999611022529271, "training/grad_norm": 0.273471862077713, "training/positions": 1000341504, "training/nonpadding_tokens": 1000326925, "throughput/steps_per_second": 0.40920749439312515, "throughput/raw_tokens_per_second": 429084.8507351448, "throughput/peak_reserved_gib": 24.275390625}
|
| 153 |
+
{"validation_step": 954, "loss": 7.030140509605408, "reconstruction_loss": 0.0, "vq_loss": 0.0, "ncp_loss": 7.030140509605408, "mstok_loss": 0.0, "semantic_loss": 0.0, "ncp_accuracy": 17.64464882943144}
|
| 154 |
+
wandb: - 0.022 MB of 0.022 MB uploaded
|
| 155 |
+
wandb:
|
| 156 |
+
wandb: Run summary:
|
| 157 |
+
wandb: throughput/peak_reserved_gib 24.27539
|
| 158 |
+
wandb: throughput/raw_tokens_per_second 429084.85074
|
| 159 |
+
wandb: throughput/steps_per_second 0.40921
|
| 160 |
+
wandb: training/generator_lr 0.0005
|
| 161 |
+
wandb: training/grad_norm 0.27347
|
| 162 |
+
wandb: training/loss 8.26384
|
| 163 |
+
wandb: training/mstok_loss 0.0
|
| 164 |
+
wandb: training/ncp_accuracy 7.03709
|
| 165 |
+
wandb: training/ncp_loss 8.26384
|
| 166 |
+
wandb: training/nonpadding_tokens 1000326925
|
| 167 |
+
wandb: training/positions 1000341504
|
| 168 |
+
wandb: training/reconstruction_loss 0.0
|
| 169 |
+
wandb: training/semantic_loss 0.0
|
| 170 |
+
wandb: training/vq_loss 0.0
|
| 171 |
+
wandb: validation/loss 7.03014
|
| 172 |
+
wandb: validation/mstok_loss 0.0
|
| 173 |
+
wandb: validation/ncp_accuracy 17.64465
|
| 174 |
+
wandb: validation/ncp_loss 7.03014
|
| 175 |
+
wandb: validation/reconstruction_loss 0.0
|
| 176 |
+
wandb: validation/semantic_loss 0.0
|
| 177 |
+
wandb: validation/vq_loss 0.0
|
| 178 |
+
wandb:
|
| 179 |
+
wandb: View run iclr-debug-two-stage-135b-20260916-generator at: https://wandb.ai/mstok/iclr-debug/runs/iclr-debug-two-stage-135b-20260916-generator
|
| 180 |
+
wandb: View project at: https://wandb.ai/mstok/iclr-debug
|
| 181 |
+
wandb: Synced 6 W&B file(s), 0 media file(s), 2 artifact file(s) and 0 other file(s)
|
| 182 |
+
wandb: Find logs at: ./wandb/run-20260916_151853-iclr-debug-two-stage-135b-20260916-generator/logs
|
| 183 |
+
wandb: WARNING wandb version 0.30.0 is available! To upgrade, please run:
|
| 184 |
+
wandb: WARNING $ pip install wandb --upgrade
|
| 185 |
+
wandb: WARNING The new W&B backend becomes opt-out in version 0.18.0; try it out with `wandb.require("core")`! See https://wandb.me/wandb-core for more information.
|
iclr-debug-two-stage-135b-20260916/generator/stopped-20260917/validation-curve.json
ADDED
|
@@ -0,0 +1,492 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"validation_step": 954,
|
| 4 |
+
"loss": 7.030140509605408,
|
| 5 |
+
"reconstruction_loss": 0.0,
|
| 6 |
+
"vq_loss": 0.0,
|
| 7 |
+
"ncp_loss": 7.030140509605408,
|
| 8 |
+
"mstok_loss": 0.0,
|
| 9 |
+
"semantic_loss": 0.0,
|
| 10 |
+
"ncp_accuracy": 17.64464882943144
|
| 11 |
+
},
|
| 12 |
+
{
|
| 13 |
+
"validation_step": 1000,
|
| 14 |
+
"loss": 6.977545759677887,
|
| 15 |
+
"reconstruction_loss": 0.0,
|
| 16 |
+
"vq_loss": 0.0,
|
| 17 |
+
"ncp_loss": 6.977545759677887,
|
| 18 |
+
"mstok_loss": 0.0,
|
| 19 |
+
"semantic_loss": 0.0,
|
| 20 |
+
"ncp_accuracy": 18.076421404682275
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"validation_step": 2000,
|
| 24 |
+
"loss": 6.475357372760773,
|
| 25 |
+
"reconstruction_loss": 0.0,
|
| 26 |
+
"vq_loss": 0.0,
|
| 27 |
+
"ncp_loss": 6.475357372760773,
|
| 28 |
+
"mstok_loss": 0.0,
|
| 29 |
+
"semantic_loss": 0.0,
|
| 30 |
+
"ncp_accuracy": 23.616304347826087
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"validation_step": 3000,
|
| 34 |
+
"loss": 6.249391813278199,
|
| 35 |
+
"reconstruction_loss": 0.0,
|
| 36 |
+
"vq_loss": 0.0,
|
| 37 |
+
"ncp_loss": 6.249391813278199,
|
| 38 |
+
"mstok_loss": 0.0,
|
| 39 |
+
"semantic_loss": 0.0,
|
| 40 |
+
"ncp_accuracy": 26.47984949832776
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"validation_step": 4000,
|
| 44 |
+
"loss": 6.114818637371063,
|
| 45 |
+
"reconstruction_loss": 0.0,
|
| 46 |
+
"vq_loss": 0.0,
|
| 47 |
+
"ncp_loss": 6.114818637371063,
|
| 48 |
+
"mstok_loss": 0.0,
|
| 49 |
+
"semantic_loss": 0.0,
|
| 50 |
+
"ncp_accuracy": 28.07140468227425
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
"validation_step": 4769,
|
| 54 |
+
"loss": 6.046309206485748,
|
| 55 |
+
"reconstruction_loss": 0.0,
|
| 56 |
+
"vq_loss": 0.0,
|
| 57 |
+
"ncp_loss": 6.046309206485748,
|
| 58 |
+
"mstok_loss": 0.0,
|
| 59 |
+
"semantic_loss": 0.0,
|
| 60 |
+
"ncp_accuracy": 28.982274247491638
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"validation_step": 5000,
|
| 64 |
+
"loss": 6.033136506080628,
|
| 65 |
+
"reconstruction_loss": 0.0,
|
| 66 |
+
"vq_loss": 0.0,
|
| 67 |
+
"ncp_loss": 6.033136506080628,
|
| 68 |
+
"mstok_loss": 0.0,
|
| 69 |
+
"semantic_loss": 0.0,
|
| 70 |
+
"ncp_accuracy": 29.018478260869564
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
"validation_step": 6000,
|
| 74 |
+
"loss": 5.966629896163941,
|
| 75 |
+
"reconstruction_loss": 0.0,
|
| 76 |
+
"vq_loss": 0.0,
|
| 77 |
+
"ncp_loss": 5.966629896163941,
|
| 78 |
+
"mstok_loss": 0.0,
|
| 79 |
+
"semantic_loss": 0.0,
|
| 80 |
+
"ncp_accuracy": 29.943729096989966
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"validation_step": 7000,
|
| 84 |
+
"loss": 5.918366656303406,
|
| 85 |
+
"reconstruction_loss": 0.0,
|
| 86 |
+
"vq_loss": 0.0,
|
| 87 |
+
"ncp_loss": 5.918366656303406,
|
| 88 |
+
"mstok_loss": 0.0,
|
| 89 |
+
"semantic_loss": 0.0,
|
| 90 |
+
"ncp_accuracy": 30.4994983277592
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"validation_step": 8000,
|
| 94 |
+
"loss": 5.8785524606704715,
|
| 95 |
+
"reconstruction_loss": 0.0,
|
| 96 |
+
"vq_loss": 0.0,
|
| 97 |
+
"ncp_loss": 5.8785524606704715,
|
| 98 |
+
"mstok_loss": 0.0,
|
| 99 |
+
"semantic_loss": 0.0,
|
| 100 |
+
"ncp_accuracy": 31.012709030100336
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"validation_step": 9000,
|
| 104 |
+
"loss": 5.842954273223877,
|
| 105 |
+
"reconstruction_loss": 0.0,
|
| 106 |
+
"vq_loss": 0.0,
|
| 107 |
+
"ncp_loss": 5.842954273223877,
|
| 108 |
+
"mstok_loss": 0.0,
|
| 109 |
+
"semantic_loss": 0.0,
|
| 110 |
+
"ncp_accuracy": 31.484698996655517
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"validation_step": 10000,
|
| 114 |
+
"loss": 5.827896971702576,
|
| 115 |
+
"reconstruction_loss": 0.0,
|
| 116 |
+
"vq_loss": 0.0,
|
| 117 |
+
"ncp_loss": 5.827896971702576,
|
| 118 |
+
"mstok_loss": 0.0,
|
| 119 |
+
"semantic_loss": 0.0,
|
| 120 |
+
"ncp_accuracy": 31.649916387959866
|
| 121 |
+
},
|
| 122 |
+
{
|
| 123 |
+
"validation_step": 11000,
|
| 124 |
+
"loss": 5.794131059646607,
|
| 125 |
+
"reconstruction_loss": 0.0,
|
| 126 |
+
"vq_loss": 0.0,
|
| 127 |
+
"ncp_loss": 5.794131059646607,
|
| 128 |
+
"mstok_loss": 0.0,
|
| 129 |
+
"semantic_loss": 0.0,
|
| 130 |
+
"ncp_accuracy": 32.15785953177257
|
| 131 |
+
},
|
| 132 |
+
{
|
| 133 |
+
"validation_step": 12000,
|
| 134 |
+
"loss": 5.785179510116577,
|
| 135 |
+
"reconstruction_loss": 0.0,
|
| 136 |
+
"vq_loss": 0.0,
|
| 137 |
+
"ncp_loss": 5.785179510116577,
|
| 138 |
+
"mstok_loss": 0.0,
|
| 139 |
+
"semantic_loss": 0.0,
|
| 140 |
+
"ncp_accuracy": 32.229933110367895
|
| 141 |
+
},
|
| 142 |
+
{
|
| 143 |
+
"validation_step": 12875,
|
| 144 |
+
"loss": 5.766966290473938,
|
| 145 |
+
"reconstruction_loss": 0.0,
|
| 146 |
+
"vq_loss": 0.0,
|
| 147 |
+
"ncp_loss": 5.766966290473938,
|
| 148 |
+
"mstok_loss": 0.0,
|
| 149 |
+
"semantic_loss": 0.0,
|
| 150 |
+
"ncp_accuracy": 32.42675585284281
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"validation_step": 13000,
|
| 154 |
+
"loss": 5.763078744411469,
|
| 155 |
+
"reconstruction_loss": 0.0,
|
| 156 |
+
"vq_loss": 0.0,
|
| 157 |
+
"ncp_loss": 5.763078744411469,
|
| 158 |
+
"mstok_loss": 0.0,
|
| 159 |
+
"semantic_loss": 0.0,
|
| 160 |
+
"ncp_accuracy": 32.53210702341137
|
| 161 |
+
},
|
| 162 |
+
{
|
| 163 |
+
"validation_step": 14000,
|
| 164 |
+
"loss": 5.748985006809234,
|
| 165 |
+
"reconstruction_loss": 0.0,
|
| 166 |
+
"vq_loss": 0.0,
|
| 167 |
+
"ncp_loss": 5.748985006809234,
|
| 168 |
+
"mstok_loss": 0.0,
|
| 169 |
+
"semantic_loss": 0.0,
|
| 170 |
+
"ncp_accuracy": 32.768645484949836
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"validation_step": 15000,
|
| 174 |
+
"loss": 5.741684236526489,
|
| 175 |
+
"reconstruction_loss": 0.0,
|
| 176 |
+
"vq_loss": 0.0,
|
| 177 |
+
"ncp_loss": 5.741684236526489,
|
| 178 |
+
"mstok_loss": 0.0,
|
| 179 |
+
"semantic_loss": 0.0,
|
| 180 |
+
"ncp_accuracy": 32.87098662207358
|
| 181 |
+
},
|
| 182 |
+
{
|
| 183 |
+
"validation_step": 16000,
|
| 184 |
+
"loss": 5.729868891239167,
|
| 185 |
+
"reconstruction_loss": 0.0,
|
| 186 |
+
"vq_loss": 0.0,
|
| 187 |
+
"ncp_loss": 5.729868891239167,
|
| 188 |
+
"mstok_loss": 0.0,
|
| 189 |
+
"semantic_loss": 0.0,
|
| 190 |
+
"ncp_accuracy": 32.96479933110368
|
| 191 |
+
},
|
| 192 |
+
{
|
| 193 |
+
"validation_step": 17000,
|
| 194 |
+
"loss": 5.731239960193634,
|
| 195 |
+
"reconstruction_loss": 0.0,
|
| 196 |
+
"vq_loss": 0.0,
|
| 197 |
+
"ncp_loss": 5.731239960193634,
|
| 198 |
+
"mstok_loss": 0.0,
|
| 199 |
+
"semantic_loss": 0.0,
|
| 200 |
+
"ncp_accuracy": 33.01923076923077
|
| 201 |
+
},
|
| 202 |
+
{
|
| 203 |
+
"validation_step": 18000,
|
| 204 |
+
"loss": 5.706959030628204,
|
| 205 |
+
"reconstruction_loss": 0.0,
|
| 206 |
+
"vq_loss": 0.0,
|
| 207 |
+
"ncp_loss": 5.706959030628204,
|
| 208 |
+
"mstok_loss": 0.0,
|
| 209 |
+
"semantic_loss": 0.0,
|
| 210 |
+
"ncp_accuracy": 33.36329431438127
|
| 211 |
+
},
|
| 212 |
+
{
|
| 213 |
+
"validation_step": 19000,
|
| 214 |
+
"loss": 5.7004749608039855,
|
| 215 |
+
"reconstruction_loss": 0.0,
|
| 216 |
+
"vq_loss": 0.0,
|
| 217 |
+
"ncp_loss": 5.7004749608039855,
|
| 218 |
+
"mstok_loss": 0.0,
|
| 219 |
+
"semantic_loss": 0.0,
|
| 220 |
+
"ncp_accuracy": 33.474916387959865
|
| 221 |
+
},
|
| 222 |
+
{
|
| 223 |
+
"validation_step": 20000,
|
| 224 |
+
"loss": 5.6975271821022035,
|
| 225 |
+
"reconstruction_loss": 0.0,
|
| 226 |
+
"vq_loss": 0.0,
|
| 227 |
+
"ncp_loss": 5.6975271821022035,
|
| 228 |
+
"mstok_loss": 0.0,
|
| 229 |
+
"semantic_loss": 0.0,
|
| 230 |
+
"ncp_accuracy": 33.45827759197324
|
| 231 |
+
},
|
| 232 |
+
{
|
| 233 |
+
"validation_step": 21000,
|
| 234 |
+
"loss": 5.6820241522789,
|
| 235 |
+
"reconstruction_loss": 0.0,
|
| 236 |
+
"vq_loss": 0.0,
|
| 237 |
+
"ncp_loss": 5.6820241522789,
|
| 238 |
+
"mstok_loss": 0.0,
|
| 239 |
+
"semantic_loss": 0.0,
|
| 240 |
+
"ncp_accuracy": 33.54824414715719
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"validation_step": 22000,
|
| 244 |
+
"loss": 5.676759459972382,
|
| 245 |
+
"reconstruction_loss": 0.0,
|
| 246 |
+
"vq_loss": 0.0,
|
| 247 |
+
"ncp_loss": 5.676759459972382,
|
| 248 |
+
"mstok_loss": 0.0,
|
| 249 |
+
"semantic_loss": 0.0,
|
| 250 |
+
"ncp_accuracy": 33.63929765886287
|
| 251 |
+
},
|
| 252 |
+
{
|
| 253 |
+
"validation_step": 23000,
|
| 254 |
+
"loss": 5.670730285644531,
|
| 255 |
+
"reconstruction_loss": 0.0,
|
| 256 |
+
"vq_loss": 0.0,
|
| 257 |
+
"ncp_loss": 5.670730285644531,
|
| 258 |
+
"mstok_loss": 0.0,
|
| 259 |
+
"semantic_loss": 0.0,
|
| 260 |
+
"ncp_accuracy": 33.70091973244147
|
| 261 |
+
},
|
| 262 |
+
{
|
| 263 |
+
"validation_step": 24000,
|
| 264 |
+
"loss": 5.658136134147644,
|
| 265 |
+
"reconstruction_loss": 0.0,
|
| 266 |
+
"vq_loss": 0.0,
|
| 267 |
+
"ncp_loss": 5.658136134147644,
|
| 268 |
+
"mstok_loss": 0.0,
|
| 269 |
+
"semantic_loss": 0.0,
|
| 270 |
+
"ncp_accuracy": 33.9554347826087
|
| 271 |
+
},
|
| 272 |
+
{
|
| 273 |
+
"validation_step": 25000,
|
| 274 |
+
"loss": 5.6540408921241765,
|
| 275 |
+
"reconstruction_loss": 0.0,
|
| 276 |
+
"vq_loss": 0.0,
|
| 277 |
+
"ncp_loss": 5.6540408921241765,
|
| 278 |
+
"mstok_loss": 0.0,
|
| 279 |
+
"semantic_loss": 0.0,
|
| 280 |
+
"ncp_accuracy": 33.90209030100335
|
| 281 |
+
},
|
| 282 |
+
{
|
| 283 |
+
"validation_step": 25750,
|
| 284 |
+
"loss": 5.651979362964631,
|
| 285 |
+
"reconstruction_loss": 0.0,
|
| 286 |
+
"vq_loss": 0.0,
|
| 287 |
+
"ncp_loss": 5.651979362964631,
|
| 288 |
+
"mstok_loss": 0.0,
|
| 289 |
+
"semantic_loss": 0.0,
|
| 290 |
+
"ncp_accuracy": 33.96053511705686
|
| 291 |
+
},
|
| 292 |
+
{
|
| 293 |
+
"validation_step": 26000,
|
| 294 |
+
"loss": 5.649268729686737,
|
| 295 |
+
"reconstruction_loss": 0.0,
|
| 296 |
+
"vq_loss": 0.0,
|
| 297 |
+
"ncp_loss": 5.649268729686737,
|
| 298 |
+
"mstok_loss": 0.0,
|
| 299 |
+
"semantic_loss": 0.0,
|
| 300 |
+
"ncp_accuracy": 33.95083612040134
|
| 301 |
+
},
|
| 302 |
+
{
|
| 303 |
+
"validation_step": 27000,
|
| 304 |
+
"loss": 5.640766968727112,
|
| 305 |
+
"reconstruction_loss": 0.0,
|
| 306 |
+
"vq_loss": 0.0,
|
| 307 |
+
"ncp_loss": 5.640766968727112,
|
| 308 |
+
"mstok_loss": 0.0,
|
| 309 |
+
"semantic_loss": 0.0,
|
| 310 |
+
"ncp_accuracy": 34.143478260869564
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"validation_step": 28000,
|
| 314 |
+
"loss": 5.643126902580262,
|
| 315 |
+
"reconstruction_loss": 0.0,
|
| 316 |
+
"vq_loss": 0.0,
|
| 317 |
+
"ncp_loss": 5.643126902580262,
|
| 318 |
+
"mstok_loss": 0.0,
|
| 319 |
+
"semantic_loss": 0.0,
|
| 320 |
+
"ncp_accuracy": 34.08010033444816
|
| 321 |
+
},
|
| 322 |
+
{
|
| 323 |
+
"validation_step": 29000,
|
| 324 |
+
"loss": 5.642015235424042,
|
| 325 |
+
"reconstruction_loss": 0.0,
|
| 326 |
+
"vq_loss": 0.0,
|
| 327 |
+
"ncp_loss": 5.642015235424042,
|
| 328 |
+
"mstok_loss": 0.0,
|
| 329 |
+
"semantic_loss": 0.0,
|
| 330 |
+
"ncp_accuracy": 34.197909698996654
|
| 331 |
+
},
|
| 332 |
+
{
|
| 333 |
+
"validation_step": 30000,
|
| 334 |
+
"loss": 5.632075030803681,
|
| 335 |
+
"reconstruction_loss": 0.0,
|
| 336 |
+
"vq_loss": 0.0,
|
| 337 |
+
"ncp_loss": 5.632075030803681,
|
| 338 |
+
"mstok_loss": 0.0,
|
| 339 |
+
"semantic_loss": 0.0,
|
| 340 |
+
"ncp_accuracy": 34.23210702341137
|
| 341 |
+
},
|
| 342 |
+
{
|
| 343 |
+
"validation_step": 31000,
|
| 344 |
+
"loss": 5.632021889686585,
|
| 345 |
+
"reconstruction_loss": 0.0,
|
| 346 |
+
"vq_loss": 0.0,
|
| 347 |
+
"ncp_loss": 5.632021889686585,
|
| 348 |
+
"mstok_loss": 0.0,
|
| 349 |
+
"semantic_loss": 0.0,
|
| 350 |
+
"ncp_accuracy": 34.24314381270903
|
| 351 |
+
},
|
| 352 |
+
{
|
| 353 |
+
"validation_step": 32000,
|
| 354 |
+
"loss": 5.6199514770507815,
|
| 355 |
+
"reconstruction_loss": 0.0,
|
| 356 |
+
"vq_loss": 0.0,
|
| 357 |
+
"ncp_loss": 5.6199514770507815,
|
| 358 |
+
"mstok_loss": 0.0,
|
| 359 |
+
"semantic_loss": 0.0,
|
| 360 |
+
"ncp_accuracy": 34.39958193979933
|
| 361 |
+
},
|
| 362 |
+
{
|
| 363 |
+
"validation_step": 33000,
|
| 364 |
+
"loss": 5.614811289310455,
|
| 365 |
+
"reconstruction_loss": 0.0,
|
| 366 |
+
"vq_loss": 0.0,
|
| 367 |
+
"ncp_loss": 5.614811289310455,
|
| 368 |
+
"mstok_loss": 0.0,
|
| 369 |
+
"semantic_loss": 0.0,
|
| 370 |
+
"ncp_accuracy": 34.48377926421405
|
| 371 |
+
},
|
| 372 |
+
{
|
| 373 |
+
"validation_step": 34000,
|
| 374 |
+
"loss": 5.617471935749054,
|
| 375 |
+
"reconstruction_loss": 0.0,
|
| 376 |
+
"vq_loss": 0.0,
|
| 377 |
+
"ncp_loss": 5.617471935749054,
|
| 378 |
+
"mstok_loss": 0.0,
|
| 379 |
+
"semantic_loss": 0.0,
|
| 380 |
+
"ncp_accuracy": 34.44857859531773
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"validation_step": 35000,
|
| 384 |
+
"loss": 5.614271013736725,
|
| 385 |
+
"reconstruction_loss": 0.0,
|
| 386 |
+
"vq_loss": 0.0,
|
| 387 |
+
"ncp_loss": 5.614271013736725,
|
| 388 |
+
"mstok_loss": 0.0,
|
| 389 |
+
"semantic_loss": 0.0,
|
| 390 |
+
"ncp_accuracy": 34.45108695652174
|
| 391 |
+
},
|
| 392 |
+
{
|
| 393 |
+
"validation_step": 36000,
|
| 394 |
+
"loss": 5.609555795192718,
|
| 395 |
+
"reconstruction_loss": 0.0,
|
| 396 |
+
"vq_loss": 0.0,
|
| 397 |
+
"ncp_loss": 5.609555795192718,
|
| 398 |
+
"mstok_loss": 0.0,
|
| 399 |
+
"semantic_loss": 0.0,
|
| 400 |
+
"ncp_accuracy": 34.55819397993311
|
| 401 |
+
},
|
| 402 |
+
{
|
| 403 |
+
"validation_step": 37000,
|
| 404 |
+
"loss": 5.612579424381257,
|
| 405 |
+
"reconstruction_loss": 0.0,
|
| 406 |
+
"vq_loss": 0.0,
|
| 407 |
+
"ncp_loss": 5.612579424381257,
|
| 408 |
+
"mstok_loss": 0.0,
|
| 409 |
+
"semantic_loss": 0.0,
|
| 410 |
+
"ncp_accuracy": 34.4626254180602
|
| 411 |
+
},
|
| 412 |
+
{
|
| 413 |
+
"validation_step": 38000,
|
| 414 |
+
"loss": 5.6007399559021,
|
| 415 |
+
"reconstruction_loss": 0.0,
|
| 416 |
+
"vq_loss": 0.0,
|
| 417 |
+
"ncp_loss": 5.6007399559021,
|
| 418 |
+
"mstok_loss": 0.0,
|
| 419 |
+
"semantic_loss": 0.0,
|
| 420 |
+
"ncp_accuracy": 34.687207357859535
|
| 421 |
+
},
|
| 422 |
+
{
|
| 423 |
+
"validation_step": 39000,
|
| 424 |
+
"loss": 5.611569912433624,
|
| 425 |
+
"reconstruction_loss": 0.0,
|
| 426 |
+
"vq_loss": 0.0,
|
| 427 |
+
"ncp_loss": 5.611569912433624,
|
| 428 |
+
"mstok_loss": 0.0,
|
| 429 |
+
"semantic_loss": 0.0,
|
| 430 |
+
"ncp_accuracy": 34.53386287625418
|
| 431 |
+
},
|
| 432 |
+
{
|
| 433 |
+
"validation_step": 40000,
|
| 434 |
+
"loss": 5.595582022666931,
|
| 435 |
+
"reconstruction_loss": 0.0,
|
| 436 |
+
"vq_loss": 0.0,
|
| 437 |
+
"ncp_loss": 5.595582022666931,
|
| 438 |
+
"mstok_loss": 0.0,
|
| 439 |
+
"semantic_loss": 0.0,
|
| 440 |
+
"ncp_accuracy": 34.729180602006686
|
| 441 |
+
},
|
| 442 |
+
{
|
| 443 |
+
"validation_step": 41000,
|
| 444 |
+
"loss": 5.603518695831299,
|
| 445 |
+
"reconstruction_loss": 0.0,
|
| 446 |
+
"vq_loss": 0.0,
|
| 447 |
+
"ncp_loss": 5.603518695831299,
|
| 448 |
+
"mstok_loss": 0.0,
|
| 449 |
+
"semantic_loss": 0.0,
|
| 450 |
+
"ncp_accuracy": 34.72182274247491
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"validation_step": 42000,
|
| 454 |
+
"loss": 5.593213293552399,
|
| 455 |
+
"reconstruction_loss": 0.0,
|
| 456 |
+
"vq_loss": 0.0,
|
| 457 |
+
"ncp_loss": 5.593213293552399,
|
| 458 |
+
"mstok_loss": 0.0,
|
| 459 |
+
"semantic_loss": 0.0,
|
| 460 |
+
"ncp_accuracy": 34.79916387959866
|
| 461 |
+
},
|
| 462 |
+
{
|
| 463 |
+
"validation_step": 43000,
|
| 464 |
+
"loss": 5.594714949131012,
|
| 465 |
+
"reconstruction_loss": 0.0,
|
| 466 |
+
"vq_loss": 0.0,
|
| 467 |
+
"ncp_loss": 5.594714949131012,
|
| 468 |
+
"mstok_loss": 0.0,
|
| 469 |
+
"semantic_loss": 0.0,
|
| 470 |
+
"ncp_accuracy": 34.682441471571906
|
| 471 |
+
},
|
| 472 |
+
{
|
| 473 |
+
"validation_step": 44000,
|
| 474 |
+
"loss": 5.592791588306427,
|
| 475 |
+
"reconstruction_loss": 0.0,
|
| 476 |
+
"vq_loss": 0.0,
|
| 477 |
+
"ncp_loss": 5.592791588306427,
|
| 478 |
+
"mstok_loss": 0.0,
|
| 479 |
+
"semantic_loss": 0.0,
|
| 480 |
+
"ncp_accuracy": 34.8190635451505
|
| 481 |
+
},
|
| 482 |
+
{
|
| 483 |
+
"validation_step": 45000,
|
| 484 |
+
"loss": 5.5970091938972475,
|
| 485 |
+
"reconstruction_loss": 0.0,
|
| 486 |
+
"vq_loss": 0.0,
|
| 487 |
+
"ncp_loss": 5.5970091938972475,
|
| 488 |
+
"mstok_loss": 0.0,
|
| 489 |
+
"semantic_loss": 0.0,
|
| 490 |
+
"ncp_accuracy": 34.62951505016722
|
| 491 |
+
}
|
| 492 |
+
]
|