ruhai-lin commited on
Commit
2ccedbf
·
verified ·
1 Parent(s): 12ee9a3

upload MixerLoop weights and CORE v2 results

Browse files
mixerloop-13m/climbmix-10B-s2026/.gitkeep DELETED
File without changes
mixerloop-13m/climbmix-10B-s2026/config.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "MixerLoopForCausalLM"
4
+ ],
5
+ "attn_mode": "chunk",
6
+ "bos_token_id": 1,
7
+ "conv_size": 4,
8
+ "dtype": "float32",
9
+ "eos_token_id": 2,
10
+ "expand_v": 2,
11
+ "fuse_cross_entropy": true,
12
+ "fuse_linear_cross_entropy": false,
13
+ "fuse_norm": true,
14
+ "fuse_swiglu": true,
15
+ "head_dim": 32,
16
+ "hidden_act": "swish",
17
+ "hidden_ratio": 4,
18
+ "hidden_size": 256,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 704,
21
+ "loop_count": 4,
22
+ "max_position_embeddings": 1024,
23
+ "model_type": "mixerloop",
24
+ "norm_eps": 1e-05,
25
+ "num_heads": 6,
26
+ "num_hidden_layers": 5,
27
+ "transformers_version": "4.57.3",
28
+ "use_cache": false,
29
+ "vocab_size": 32000
30
+ }
mixerloop-13m/climbmix-10B-s2026/eval/core_eval.csv ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Task,Accuracy,Centered
2
+ hellaswag_zeroshot,0.276539,0.035385
3
+ jeopardy,0.000000,0.000000
4
+ bigbench_qa_wikidata,0.016879,0.016879
5
+ arc_easy,0.372475,0.163300
6
+ arc_challenge,0.208191,-0.055745
7
+ copa,0.460000,-0.080000
8
+ commonsense_qa,0.209664,-0.323846
9
+ piqa,0.590860,0.181719
10
+ openbook_qa,0.300000,0.066667
11
+ lambada_openai,0.125364,0.125364
12
+ hellaswag,0.275742,0.034323
13
+ winograd,0.505495,0.010989
14
+ winogrande,0.518548,0.037096
15
+ bigbench_dyck_languages,0.018000,0.018000
16
+ agi_eval_lsat_ar,0.186957,-0.084058
17
+ bigbench_cs_algorithms,0.267424,0.267424
18
+ bigbench_operators,0.119048,0.119048
19
+ bigbench_repeat_copy_logic,0.000000,0.000000
20
+ squad,0.012015,0.012015
21
+ coqa,0.041087,0.041087
22
+ boolq,0.524771,-0.250604
23
+ bigbench_language_identification,0.253900,0.005200
24
+ Core_v2,,0.015466
mixerloop-13m/climbmix-10B-s2026/generation_config.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 1,
4
+ "eos_token_id": 2,
5
+ "transformers_version": "4.57.3",
6
+ "use_cache": false
7
+ }
mixerloop-13m/climbmix-10B-s2026/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bd34ff6d82284ca5687b009ce3fb46f4714624284631dc8fc7e00670776e8218
3
+ size 51595152
mixerloop-13m/climbmix-10B-s2026/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<s>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "</s>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "unk_token": {
17
+ "content": "<unk>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
mixerloop-13m/climbmix-10B-s2026/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
mixerloop-13m/climbmix-10B-s2026/tokenizer.model ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
3
+ size 499723
mixerloop-13m/climbmix-10B-s2026/tokenizer_config.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "0": {
7
+ "content": "<unk>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "1": {
15
+ "content": "<s>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": true
21
+ },
22
+ "2": {
23
+ "content": "</s>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": true
29
+ }
30
+ },
31
+ "bos_token": "<s>",
32
+ "clean_up_tokenization_spaces": false,
33
+ "eos_token": "</s>",
34
+ "extra_special_tokens": {},
35
+ "legacy": true,
36
+ "model_max_length": 1024,
37
+ "pad_token": null,
38
+ "sp_model_kwargs": {},
39
+ "spaces_between_special_tokens": false,
40
+ "tokenizer_class": "LlamaTokenizer",
41
+ "unk_token": "<unk>",
42
+ "use_default_system_prompt": false
43
+ }
mixerloop-13m/climbmix-10B-s42/.gitkeep DELETED
File without changes
mixerloop-13m/climbmix-10B-s42/config.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "MixerLoopForCausalLM"
4
+ ],
5
+ "attn_mode": "chunk",
6
+ "bos_token_id": 1,
7
+ "conv_size": 4,
8
+ "dtype": "float32",
9
+ "eos_token_id": 2,
10
+ "expand_v": 2,
11
+ "fuse_cross_entropy": true,
12
+ "fuse_linear_cross_entropy": false,
13
+ "fuse_norm": true,
14
+ "fuse_swiglu": true,
15
+ "head_dim": 32,
16
+ "hidden_act": "swish",
17
+ "hidden_ratio": 4,
18
+ "hidden_size": 256,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 704,
21
+ "loop_count": 4,
22
+ "max_position_embeddings": 1024,
23
+ "model_type": "mixerloop",
24
+ "norm_eps": 1e-05,
25
+ "num_heads": 6,
26
+ "num_hidden_layers": 5,
27
+ "transformers_version": "4.57.3",
28
+ "use_cache": false,
29
+ "vocab_size": 32000
30
+ }
mixerloop-13m/climbmix-10B-s42/eval/core_eval.csv ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Task,Accuracy,Centered
2
+ hellaswag_zeroshot,0.274248,0.032331
3
+ jeopardy,0.000472,0.000472
4
+ bigbench_qa_wikidata,0.026967,0.026967
5
+ arc_easy,0.354377,0.139169
6
+ arc_challenge,0.209898,-0.053470
7
+ copa,0.460000,-0.080000
8
+ commonsense_qa,0.312039,-0.152363
9
+ piqa,0.573993,0.147987
10
+ openbook_qa,0.280000,0.040000
11
+ lambada_openai,0.114690,0.114690
12
+ hellaswag,0.271559,0.028746
13
+ winograd,0.531136,0.062271
14
+ winogrande,0.490134,-0.019732
15
+ bigbench_dyck_languages,0.002000,0.002000
16
+ agi_eval_lsat_ar,0.273913,0.031884
17
+ bigbench_cs_algorithms,0.339394,0.339394
18
+ bigbench_operators,0.147619,0.147619
19
+ bigbench_repeat_copy_logic,0.000000,0.000000
20
+ squad,0.003500,0.003500
21
+ coqa,0.030064,0.030064
22
+ boolq,0.380428,-0.630452
23
+ bigbench_language_identification,0.257400,0.009867
24
+ Core_v2,,0.010043
mixerloop-13m/climbmix-10B-s42/generation_config.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 1,
4
+ "eos_token_id": 2,
5
+ "transformers_version": "4.57.3",
6
+ "use_cache": false
7
+ }
mixerloop-13m/climbmix-10B-s42/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4c20e24242cb9a76ff7c543a3369ceddd08f82e92ef7bce04a4a0662c822d684
3
+ size 51595152
mixerloop-13m/climbmix-10B-s42/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<s>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "</s>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "unk_token": {
17
+ "content": "<unk>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
mixerloop-13m/climbmix-10B-s42/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
mixerloop-13m/climbmix-10B-s42/tokenizer.model ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
3
+ size 499723
mixerloop-13m/climbmix-10B-s42/tokenizer_config.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "0": {
7
+ "content": "<unk>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "1": {
15
+ "content": "<s>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": true
21
+ },
22
+ "2": {
23
+ "content": "</s>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": true
29
+ }
30
+ },
31
+ "bos_token": "<s>",
32
+ "clean_up_tokenization_spaces": false,
33
+ "eos_token": "</s>",
34
+ "extra_special_tokens": {},
35
+ "legacy": true,
36
+ "model_max_length": 1024,
37
+ "pad_token": null,
38
+ "sp_model_kwargs": {},
39
+ "spaces_between_special_tokens": false,
40
+ "tokenizer_class": "LlamaTokenizer",
41
+ "unk_token": "<unk>",
42
+ "use_default_system_prompt": false
43
+ }