M31-Entropy pretrain 500
Browse files- experiments/M31-Entropy/checkpoints/pretrain/step-00000500/README.md +1 -0
- experiments/M31-Entropy/checkpoints/pretrain/step-00000500/checkpoint.json +11 -0
- experiments/M31-Entropy/checkpoints/pretrain/step-00000500/config.json +19 -0
- experiments/M31-Entropy/checkpoints/pretrain/step-00000500/model.safetensors +3 -0
- experiments/M31-Entropy/checkpoints/pretrain/step-00000500/tokenizer.json +0 -0
experiments/M31-Entropy/checkpoints/pretrain/step-00000500/README.md
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
# M31-Entropy-220M\n\nReasoning-first Python model trained from scratch.\n
|
experiments/M31-Entropy/checkpoints/pretrain/step-00000500/checkpoint.json
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"experiment": "M31-Entropy-220M",
|
| 3 |
+
"stage": "pretrain",
|
| 4 |
+
"step": 500,
|
| 5 |
+
"metrics": {
|
| 6 |
+
"loss": 5.152145743370056,
|
| 7 |
+
"avg50": 5.1091614651679995
|
| 8 |
+
},
|
| 9 |
+
"repo_id": "eshanized/M31Entropy",
|
| 10 |
+
"path": "experiments/M31-Entropy/checkpoints/pretrain/step-00000500"
|
| 11 |
+
}
|
experiments/M31-Entropy/checkpoints/pretrain/step-00000500/config.json
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model_type": "m31entropy",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"M31ENTROPY"
|
| 5 |
+
],
|
| 6 |
+
"vocab_size": 128000,
|
| 7 |
+
"hidden_size": 896,
|
| 8 |
+
"num_hidden_layers": 20,
|
| 9 |
+
"num_attention_heads": 14,
|
| 10 |
+
"num_key_value_heads": 7,
|
| 11 |
+
"intermediate_size": 2816,
|
| 12 |
+
"max_position_embeddings": 2048,
|
| 13 |
+
"training_sequence_length": 1024,
|
| 14 |
+
"rope_theta": 10000.0,
|
| 15 |
+
"rms_norm_eps": 1e-06,
|
| 16 |
+
"tie_word_embeddings": true,
|
| 17 |
+
"parameter_count": 314281856,
|
| 18 |
+
"experiment": "M31-Entropy-220M"
|
| 19 |
+
}
|
experiments/M31-Entropy/checkpoints/pretrain/step-00000500/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3b56b52c34e2897cd9c01be5585f3bc822c617600b2cc3db372c897da2c8e16b
|
| 3 |
+
size 1257143008
|
experiments/M31-Entropy/checkpoints/pretrain/step-00000500/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|