ospanbatyr commited on
Commit
ced5edf
·
verified ·
1 Parent(s): 92787b3

Add 12 model file(s)

Browse files
t1_pure_3d_s_eb512_lrsqrt/checkpoint-0.8Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b9e223a9e32abda6c9942880517a5f8215d83c37012852d63ca6c2aadeedb92b
3
+ size 2102190459
t1_pure_3d_s_eb512_lrsqrt/checkpoint-1.5Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cf72988afb8357dc8e0bf66827647fed7b6c63d337ab7bcc98da4d331397d005
3
+ size 2102190459
t1_pure_3d_s_eb512_lrsqrt/checkpoint-2Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eb3956b9560689c5c866b3a17e5aed6ecc1ddcb5d136675dcee7d2d51c248511
3
+ size 2102189413
t1_pure_3d_s_eb512_lrsqrt/checkpoint-3Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6c6a9a1cd20f4d610415131ec331a5537fc5b7e8e9c96cc6eea691191b96462
3
+ size 2102189413
t1_pure_3d_s_eb512_lrsqrt/checkpoint-4Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5a246f4216a57f36bf6a88430e8fde1bee51ae1ffe924fc1d5891ea6d1b60514
3
+ size 2102189413
t1_pure_3d_s_eb512_lrsqrt/checkpoint-5.2Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38842d2b62707dbbf2b9c0880356df8450ac43866eaa1eb2955a95ead8b9a6aa
3
+ size 2102190459
t1_pure_3d_s_eb512_lrsqrt/checkpoint-final.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a13a339fc3c1e453375f01e8852e1a9ba784adbbe17acb0bf60ae7393d43aaac
3
+ size 2102189936
t1_pure_3d_xs_eb512_lrfull/checkpoint-0.4Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b384e84f31c266eeeed6cd0fa25f7e42754f233dd1da4d4d75d4b8e5551eae57
3
+ size 1175134587
t1_pure_3d_xs_eb512_lrfull/checkpoint-0.8Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6226941adc562b3f73537a0b5b6f60a4acdfc06e3e074260576625eae06a71b
3
+ size 1175134587
t1_pure_3d_xs_eb512_lrfull/checkpoint-1.5Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4c3fad8e8cd99ea37707e1adca3cb62b681aad6001fb26ac59e9d2517703fc0d
3
+ size 1175134587
t1_pure_3d_xs_eb512_lrfull/checkpoint-2Blb.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a566a857f78210fe6253ba2c1bf82e310650082cbf58a0765baaf8ba17f8ea19
3
+ size 1175133541
t1_pure_3d_xs_eb512_lrfull/config.json ADDED
@@ -0,0 +1,111 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "run_name": "t1_pure_3d_xs_eb512_lrfull",
3
+ "epoch": 0,
4
+ "n_params": 100748296,
5
+ "args": {
6
+ "run_name": "t1_pure_3d_xs_eb512_lrfull",
7
+ "batch_size": 32,
8
+ "epochs": 1,
9
+ "total_tokens": -1,
10
+ "lossbearing_tokens": 5.2,
11
+ "accum_iter": 16,
12
+ "effective_batch_size": 512,
13
+ "lr_batch_scaling": "none",
14
+ "save_ckpt_freq": 1,
15
+ "save_ckpt_freq_tokens": 2,
16
+ "debug_batch_stats": 0,
17
+ "model": "fm_xs_12d_swiglu_nobias",
18
+ "dtype": "bfloat16",
19
+ "loss_type": "token",
20
+ "context_length": 2048,
21
+ "pos_emb_ablation": "vanilla",
22
+ "max_blocks": 32,
23
+ "intra_pos_emb": "default",
24
+ "video_pos_3d": false,
25
+ "finetune": "",
26
+ "opt": "adamw",
27
+ "opt_eps": 1e-08,
28
+ "opt_betas": [
29
+ 0.9,
30
+ 0.95
31
+ ],
32
+ "compute_grad_norm": true,
33
+ "clip_grad": 1.0,
34
+ "skip_grad": null,
35
+ "momentum": 0.9,
36
+ "weight_decay": 0.1,
37
+ "weight_decay_end": 0.1,
38
+ "blr": 0.001,
39
+ "min_blr": 5e-05,
40
+ "scheduler": "wsd",
41
+ "decay_fraction": 0.1,
42
+ "warmup_steps": 250,
43
+ "cooldown_steps": -1,
44
+ "data_config": "cfgs/default/4m/experiments/t1/unimodal/3d/data_train_pure.yaml",
45
+ "data_config_val": "cfgs/default/4m/experiments/t1/unimodal/3d/data_val_pure.yaml",
46
+ "cross_eval_val": "",
47
+ "s3_endpoint": "",
48
+ "s3_data_endpoint": "",
49
+ "s3_multipart_chunksize_mb": 512,
50
+ "s3_multipart_threshold_mb": 512,
51
+ "s3_max_io_queue": 100,
52
+ "text_tokenizer_path": "fourm/utils/tokenizer/trained/smollm_tokenizer.json",
53
+ "eval_freq": 1,
54
+ "eval_freq_tokens": 1,
55
+ "dist_eval": true,
56
+ "eval": false,
57
+ "val_sampling_mode": "without_replacement",
58
+ "output_dir": "output/hb/t1_pure_3d_xs_eb512_lrfull",
59
+ "device": "cuda",
60
+ "seed": 0,
61
+ "resume": "",
62
+ "auto_resume": true,
63
+ "start_epoch": 0,
64
+ "num_workers": 10,
65
+ "pin_mem": true,
66
+ "find_unused_params": true,
67
+ "rlimit": 4096,
68
+ "print_all": false,
69
+ "show_user_warnings": false,
70
+ "s3_save_dir": "",
71
+ "dist_url": "env://",
72
+ "log_wandb": true,
73
+ "wandb_project": "loom",
74
+ "wandb_entity": null,
75
+ "wandb_run_name": "t1_pure_3d_xs_eb512_lrfull",
76
+ "wandb_group": "t1-halfbatch",
77
+ "dim": null,
78
+ "depth": null,
79
+ "num_heads": null,
80
+ "mlp_ratio": null,
81
+ "log_grad_norms_every": 0,
82
+ "save_at_tokens": "0.4,0.8,1.5,3,4,5.2",
83
+ "unigram_entropies": null,
84
+ "cooldown_branch": false,
85
+ "cooldown_tokens": 0,
86
+ "config_path": "cfgs/default/4m/experiments/t1/unimodal/3d/run_xs_eb512_lrfull.yaml",
87
+ "rank": 0,
88
+ "world_size": 1,
89
+ "gpu": 0,
90
+ "distributed": true,
91
+ "dist_backend": "nccl",
92
+ "num_tasks": 1,
93
+ "sampling_mode": "without_replacement",
94
+ "max_micro_steps": 158692,
95
+ "lossbearing_target": 5200000000.0,
96
+ "lossbearing_mode": true,
97
+ "lr": 0.001,
98
+ "min_lr": 5e-05,
99
+ "global_micro_step": 158692,
100
+ "cum_lossbearing": 5200948224.0,
101
+ "cum_mod_tokens": {
102
+ "text": 5538816,
103
+ "tok_3d": 5195409408
104
+ }
105
+ },
106
+ "source_configs": {
107
+ "cfgs/default/4m/experiments/t1/unimodal/3d/run_xs_eb512_lrfull.yaml": "# Half-batch step-count CONTROL for pure 3d-XS (P0 \u2014 scripts/gen_halfbatch_configs.py).\n# Clone of run_xs_pure.yaml with ONE variable changed: effective batch 1024 -> 512,\n# so the SAME 5.2B loss-bearing tokens are consumed over ~2x the optimizer steps \u2014\n# the recipe a 50/50 bimodal cell silently gives its recipient, with NO partner.\n# LR arm 'lrfull': lr = blr \u2014 holds the pair's literal absolute per-step rate.\n# Baseline to difference against: output/lb/t1_pure_3d_xs (same seed, same D, batch 1024).\n_base_: ../../../../models/dec/4m-xs-dec.yaml\nrun_name: t1_pure_3d_xs_eb512_lrfull\ntotal_tokens: -1\nlossbearing_tokens: 5.2\ndata_config: cfgs/default/4m/experiments/t1/unimodal/3d/data_train_pure.yaml\ndata_config_val: cfgs/default/4m/experiments/t1/unimodal/3d/data_val_pure.yaml\naccum_iter: 16\neffective_batch_size: 512\nlr_batch_scaling: none\nwarmup_steps: 250\neval_freq_tokens: 1\nsave_ckpt_freq_tokens: 2\nsave_at_tokens: 0.4,0.8,1.5,3,4,5.2\nwandb_group: t1-halfbatch\n",
108
+ "cfgs/default/4m/experiments/t1/unimodal/3d/data_train_pure.yaml": "# T1 pure unimodal training data \u2014 3d\n# Text stripped from all sources (pure modality-only condition)\n\nmodalities:\n- tok_3d\nepoch_size: 50000000\nsampler_type: temperature\ntemperature_tau: 1.0\nsources:\n- name: 3d\n shards: data/shards/3d_shuffled/train/shard_{000000..000685}.tar\n weight: 685736\n modality_tokens:\n tok_3d: 1875\n format: npz\nsampling_mode: without_replacement\n",
109
+ "cfgs/default/4m/experiments/t1/unimodal/3d/data_val_pure.yaml": "# T1 pure unimodal validation data \u2014 3d\n# Text stripped from all sources (pure modality-only condition)\n\nmodalities:\n- tok_3d\nval_steps: 200\nsampler_type: uniform\nsources:\n- name: 3d_val\n shards: data/shards/3d_shuffled/val/shard_{000000..000013}.tar\n weight: 14000\n modality_tokens:\n tok_3d: 1875\n format: npz\n"
110
+ }
111
+ }