appvoid commited on
Commit
da49047
·
0 Parent(s):

Clean model history

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +35 -0
  2. README.md +82 -0
  3. bet_model.py +488 -0
  4. checkpoints/sparkbet9m/checkpoint-000000221000/COMPLETE +1 -0
  5. checkpoints/sparkbet9m/checkpoint-000000221000/metadata.json +10 -0
  6. checkpoints/sparkbet9m/checkpoint-000000221000/training.pt +3 -0
  7. checkpoints/sparkbet9m/checkpoint-000000222000/COMPLETE +1 -0
  8. checkpoints/sparkbet9m/checkpoint-000000222000/metadata.json +10 -0
  9. checkpoints/sparkbet9m/checkpoint-000000222000/training.pt +3 -0
  10. config.json +40 -0
  11. configuration_bet.py +64 -0
  12. dataset_manifest.json +0 -0
  13. inference.py +43 -0
  14. model.safetensors +3 -0
  15. modeling_bet.py +76 -0
  16. records.py +37 -0
  17. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789334934.modal.206.0.step-000000000000 +3 -0
  18. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789335534.modal.206.1.step-000000000171 +3 -0
  19. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789336136.modal.206.2.step-000000000366 +3 -0
  20. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789336741.modal.206.3.step-000000000564 +3 -0
  21. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789337343.modal.206.4.step-000000000765 +3 -0
  22. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789337947.modal.206.5.step-000000000960 +3 -0
  23. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789338549.modal.206.6.step-000000001156 +3 -0
  24. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789339151.modal.206.7.step-000000001355 +3 -0
  25. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789339756.modal.206.8.step-000000001551 +3 -0
  26. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789340360.modal.206.9.step-000000001749 +3 -0
  27. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789340961.modal.206.10.step-000000001942 +3 -0
  28. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789341565.modal.206.11.step-000000002134 +3 -0
  29. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789342170.modal.206.12.step-000000002331 +3 -0
  30. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789342775.modal.206.13.step-000000002529 +3 -0
  31. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789343379.modal.206.14.step-000000002730 +3 -0
  32. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789343984.modal.206.15.step-000000002924 +3 -0
  33. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789344588.modal.206.16.step-000000003101 +3 -0
  34. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789345191.modal.206.17.step-000000003288 +3 -0
  35. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789345794.modal.206.18.step-000000003483 +3 -0
  36. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789346396.modal.206.19.step-000000003680 +3 -0
  37. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789346999.modal.206.20.step-000000003873 +3 -0
  38. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789347602.modal.206.21.step-000000004067 +3 -0
  39. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789348204.modal.206.22.step-000000004259 +3 -0
  40. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789348809.modal.206.23.step-000000004455 +3 -0
  41. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789349415.modal.206.24.step-000000004650 +3 -0
  42. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789350017.modal.206.25.step-000000004846 +3 -0
  43. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789350620.modal.206.26.step-000000005048 +3 -0
  44. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789351224.modal.206.27.step-000000005247 +3 -0
  45. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789351828.modal.206.28.step-000000005445 +3 -0
  46. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789352433.modal.206.29.step-000000005639 +3 -0
  47. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789353038.modal.206.30.step-000000005834 +3 -0
  48. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789353643.modal.206.31.step-000000006029 +3 -0
  49. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789354246.modal.206.32.step-000000006227 +3 -0
  50. runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789354851.modal.206.33.step-000000006422 +3 -0
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,82 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: transformers
3
+ pipeline_tag: text-generation
4
+ tags: [bet, byte-level, recurrent, looped]
5
+ ---
6
+ # Cortex — SparkBET-9M
7
+
8
+ Target repository: **appvoid/cortex**. The current training model has **9,353,876 parameters**: width 324, FFN 864, one prelude block, six physical recurrent body blocks, one coda block, 6 query heads / 2 KV heads, head dimension 54, rank-16 phase LoRA, two loop-level Hyper-Connection lanes, Deep-Delta body residuals, QK normalization, and continuous phase/stride conditioning. The byte vocabulary has 259 IDs (0–255 plus PAD/BOS/EOS), context is 1,024 IDs, and the maximum recurrent budget is L8.
9
+
10
+ ## Run
11
+
12
+ Import the notebook into Kaggle (two T4 GPUs) or a Modal notebook (A10/A10G or RTX PRO 6000), enable network access, provide `HF_TOKEN`, then **Run All**. Kaggle Secrets and environment variables are both supported. The same FP16 autocast + GradScaler precision policy is used on every supported CUDA family; larger-memory GPUs spend memory on larger physical microbatches before activation checkpointing is enabled.
13
+
14
+ The trainer runs continuously until interrupted. A fresh run warms up once and then keeps a nonzero constant learning rate. Interrupting the training cell requests a clean stop and waits for a completed optimizer update, local full-state save, Hugging Face full-state upload, metrics upload, and inference export. A hard kill can recover from the most recent durable checkpoint.
15
+
16
+ All project source remains embedded in readable notebook cells and is materialized by Run All. This includes data streams, the Cortex curriculum, model, trainer, full-state recovery, Hugging Face export, tokenizer/model wrappers, provenance files, and tests. Existing codec utility modules remain packaged for project continuity, but the current training registry contains **no image or audio datasets**.
17
+
18
+ ## Active training data
19
+
20
+ The active external language sources are:
21
+
22
+ - `appvoid/rewrite6`
23
+ - Ultra-FineWeb-L3 English multi-style
24
+ - Ultra-FineWeb-L3 English QA
25
+ - DCLM baseline 1.0
26
+ - FineWeb-Edu score 2
27
+ - FineMath 4+
28
+ - `appvoid/no-prompt-oasst`
29
+ - `appvoid/no-prompt-openhermes`
30
+ - SmolLM-Corpus Cosmopedia v2
31
+ - SmolLM-Corpus FineWeb-Edu-dedup
32
+
33
+ The procedural Cortex curriculum remains active as its own weighted source. Dataset revisions and shard lists are pinned in `dataset_manifest.json`; each source keeps deterministic cursor/epoch/rejection state in full checkpoints.
34
+
35
+ ## Objective and recurrent compute
36
+
37
+ Every successful optimizer update trains the exact L8 trajectory plus one exact auxiliary trajectory, keeping the same objective:
38
+
39
+ `CE(L8) + 0.20 * CE(Lr), r in {1,...,7}`.
40
+
41
+ The auxiliary depth follows a deterministic, repeating **sustained progressive-data curriculum**, not a rapid 28-update depth rotation. With `aux_stage_base_updates=128` the L8+L1 stage receives 128 committed optimizer updates and batch draws, L8+L2 receives 256, then L3=384, L4=512, L5=640, L6=768, and L7=896; one curriculum spans 3,584 updates before repeating. Every update draws another global mixer batch (which can contain repeated finite-dataset samples after epochs, not guaranteed unique records). Hence L7 receives 7× the optimizer updates and example presentations allocated to L1, and **L8 remains supervised in every update**. Tune only `aux_stage_base_updates` to scale the entire schedule; the checkpoint persists this value and the curriculum origin. Depth is independent of microbatch count, DDP partition and overflow retries, and is reconstructed from the committed checkpoint step on resume. Each exact-budget trajectory receives its own continuous phase/stride coordinates; auxiliary losses are not snapshots from L8. Full BPTT is retained; activation checkpointing remains a GPU-memory fallback. More exposure encourages but does not guarantee monotonic validation performance.
42
+
43
+ The exact previous SparkBET notebook fingerprint and exact original pre-schedule fingerprint are explicitly approved for *verified auxiliary-schedule migration*. Migrated checkpoints begin their new data stages at L1 using their committed update as `aux_curriculum_origin_step`, while preserving model, optimizer, scaler, and data cursors. A subsequent resume restores the saved origin and stage budget instead of restarting L1. Other fingerprints are rejected rather than silently restarting. Set `allow_verified_schedule_migration=False` to forbid even these approved migrations. The new objective fingerprint is stored on the next checkpoint. BET2/LeWorldModel experimental code is included, but `BET2_ABLATION_ENABLED=False` by default; standard training does not create or train the sidecar.
44
+
45
+ ## Hardware policy
46
+
47
+ Run All identifies T4, A10/A10G, and RTX PRO 6000 explicitly. It first probes the largest divisible physical microbatch without activation checkpointing. If none fits with safety headroom, it retries with checkpointing. This keeps model/context/precision fixed while using larger-memory GPUs for larger physical batches. TF32 remains disabled so the numerical precision policy does not silently change when a run moves between supported GPU families.
48
+
49
+ ## Checkpoints, resume, and Hugging Face
50
+
51
+ Complete local checkpoint directories contain `training.pt`, `metadata.json`, and `COMPLETE`. They preserve model, AdamW, GradScaler, per-rank RNG, exact mixer/data cursors, update counters, and training contract. Local retention keeps three complete checkpoints.
52
+
53
+ Hugging Face full-state checkpoints are uploaded to a SparkBET-specific namespace inside **appvoid/cortex** so they cannot be confused with earlier architectures. Hub head retention keeps two complete SparkBET checkpoints; repository history is not rewritten. At startup the trainer compares verified local and Hub candidates newest-first and can recover across sessions. Upload failures preserve local state and are reported rather than deleting a completed checkpoint.
54
+
55
+ TensorBoard event files are uploaded independently to `runs/`; the model page exposes them under the repository TensorBoard view. A standalone Transformers export is uploaded periodically and on a clean stop. The export contains SafeTensors weights, exact configuration, byte tokenizer, model card, dataset manifest, and custom model code for `trust_remote_code=True`.
56
+
57
+ ## Hugging Face authentication
58
+
59
+ `HF_TOKEN` is read from Kaggle Secrets first on Kaggle and otherwise from the environment. It is never printed. The token needs read access to any private training source you use and write access to `appvoid/cortex`.
60
+
61
+ ## Inference
62
+
63
+ The Hub export supports:
64
+
65
+ ```python
66
+ from transformers import AutoTokenizer, AutoModelForCausalLM
67
+ import torch
68
+
69
+ repo = "appvoid/cortex"
70
+ tok = AutoTokenizer.from_pretrained(repo, trust_remote_code=True)
71
+ model = AutoModelForCausalLM.from_pretrained(repo, trust_remote_code=True).cuda().eval()
72
+ ids = torch.tensor([[257] + tok.encode("The next step is", add_special_tokens=False)], device="cuda")
73
+ with torch.inference_mode(), torch.autocast("cuda", dtype=torch.float16):
74
+ result = model.generate(ids, max_new_tokens=200, do_sample=False, use_cache=False)
75
+ print(tok.decode(result[0], skip_special_tokens=True))
76
+ ```
77
+
78
+ The current model does not implement a recurrent KV cache, so generation recomputes the retained context for each new byte.
79
+
80
+ ## Validation
81
+
82
+ The notebook runs source-integrity checks, architecture fingerprint/count checks, semantic/curriculum tests, data-stream tests, checkpoint recovery tests, Hugging Face export structure tests, and DDP equivalence checks before training. GPU capacity is validated with a real full-context L8+L7 backward/Adam preflight before the first committed update.
bet_model.py ADDED
@@ -0,0 +1,488 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """SparkBET-9M: min-spark-style looped core with BET phase conditioning."""
2
+ import math, hashlib
3
+ from dataclasses import dataclass, asdict
4
+ import torch
5
+ from torch import nn
6
+ import torch.nn.functional as F
7
+ from torch.utils.checkpoint import checkpoint
8
+
9
+ SEQ_LEN = 1024
10
+ MAX_LOOPS = 8
11
+ PAD_ID, BOS_ID, EOS_ID = 256, 257, 258
12
+ VOCAB_SIZE = 259
13
+ EXPECTED_PARAM_COUNT = 9_353_876
14
+ EXPECTED_ARCH_SHAPE_SHA256 = "c85415fc50d03a23f89ccd870ebf53a3a00c5a61017c13a28f6444fdd71b72b9"
15
+ _USE_GRAD_CHECKPOINTING = False
16
+ _STATE_NOISE_SIGMA = 0.0
17
+
18
+
19
+ def set_gradient_checkpointing(enabled):
20
+ global _USE_GRAD_CHECKPOINTING
21
+ _USE_GRAD_CHECKPOINTING = bool(enabled)
22
+
23
+
24
+ def set_state_noise_sigma(value):
25
+ global _STATE_NOISE_SIGMA
26
+ value = float(value)
27
+ if value < 0:
28
+ raise ValueError("state noise sigma must be >= 0")
29
+ _STATE_NOISE_SIGMA = value
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class BETConfig:
34
+ vocab_size: int = VOCAB_SIZE
35
+ hidden_size: int = 324
36
+ intermediate_size: int = 864
37
+ prelude_layers: int = 1
38
+ body_blocks: int = 6
39
+ coda_layers: int = 1
40
+ num_heads: int = 6
41
+ num_kv_heads: int = 2
42
+ head_dim: int = 54
43
+ lora_rank: int = 16
44
+ hyper_lanes: int = 2
45
+ max_seq_len: int = SEQ_LEN
46
+ max_loops: int = MAX_LOOPS
47
+ rope_theta: float = 10_000.0
48
+ rms_eps: float = 1e-6
49
+ ddl_beta_init: float = 1.0
50
+ ddl_k_eps: float = 1e-2
51
+ ddl_v_sigmoid_scale: float = 4.0
52
+
53
+ @property
54
+ def q_dim(self):
55
+ return self.num_heads * self.head_dim
56
+
57
+ @property
58
+ def kv_dim(self):
59
+ return self.num_kv_heads * self.head_dim
60
+
61
+ @property
62
+ def qkv_dim(self):
63
+ return self.q_dim + 2 * self.kv_dim
64
+
65
+
66
+ CFG = BETConfig()
67
+
68
+
69
+ class RMSNorm(nn.Module):
70
+ def __init__(self, dim, eps=1e-6):
71
+ super().__init__()
72
+ self.weight = nn.Parameter(torch.ones(dim))
73
+ self.eps = eps
74
+
75
+ def forward(self, x):
76
+ dtype = x.dtype
77
+ y = x.float()
78
+ y = y * torch.rsqrt(y.pow(2).mean(-1, keepdim=True) + self.eps)
79
+ return (y * self.weight.float()).to(dtype)
80
+
81
+
82
+ def rope_cos_sin(position_ids, dim, theta, dtype):
83
+ inv = 1.0 / (theta ** (torch.arange(0, dim, 2, device=position_ids.device, dtype=torch.float32) / dim))
84
+ f = position_ids.float().unsqueeze(-1) * inv
85
+ return f.cos().unsqueeze(1).to(dtype), f.sin().unsqueeze(1).to(dtype)
86
+
87
+
88
+ def apply_rope(x, cos, sin):
89
+ # RoPE tables may be prepared before the first autocast linear, when the
90
+ # embedding stream is FP32. Cast them to the projected Q/K dtype here so
91
+ # attention remains FP16 on every CUDA profile instead of being promoted.
92
+ cos, sin = cos.to(dtype=x.dtype), sin.to(dtype=x.dtype)
93
+ xe, xo = x[..., 0::2], x[..., 1::2]
94
+ return torch.stack((xe * cos - xo * sin, xe * sin + xo * cos), dim=-1).flatten(-2)
95
+
96
+
97
+ def attention_mask_and_positions(input_ids, attention_mask=None):
98
+ b, t = input_ids.shape
99
+ if attention_mask is None:
100
+ pos = torch.arange(t, device=input_ids.device).view(1, t).expand(b, t)
101
+ return None, pos
102
+ mask = attention_mask.to(device=input_ids.device, dtype=torch.bool)
103
+ if mask.shape != input_ids.shape:
104
+ raise ValueError(f"attention_mask shape {tuple(mask.shape)} != input_ids {tuple(input_ids.shape)}")
105
+ pos = (mask.long().cumsum(-1) - 1).clamp_min(0)
106
+ if bool(mask.all()):
107
+ return None, pos
108
+ causal = torch.ones((t, t), device=input_ids.device, dtype=torch.bool).tril()[None, None]
109
+ allowed = causal & mask[:, None, None, :]
110
+ return allowed, pos
111
+
112
+
113
+ class Attention(nn.Module):
114
+ """6Q/2KV GQA with per-head QK RMSNorm before RoPE."""
115
+ def __init__(self, c):
116
+ super().__init__()
117
+ if c.hidden_size != c.q_dim:
118
+ raise ValueError("hidden_size must equal num_heads * head_dim")
119
+ if c.num_heads % c.num_kv_heads:
120
+ raise ValueError("num_heads must be divisible by num_kv_heads")
121
+ self.qkv = nn.Linear(c.hidden_size, c.qkv_dim, bias=False)
122
+ self.out = nn.Linear(c.q_dim, c.hidden_size, bias=False)
123
+ self.qn = RMSNorm(c.head_dim, c.rms_eps)
124
+ self.kn = RMSNorm(c.head_dim, c.rms_eps)
125
+ self.nh, self.nkv, self.hd = c.num_heads, c.num_kv_heads, c.head_dim
126
+ self.q_dim, self.kv_dim = c.q_dim, c.kv_dim
127
+
128
+ def forward(self, x, cos, sin, qkv_delta=None, attn_mask=None):
129
+ b, t, _ = x.shape
130
+ qkv = self.qkv(x)
131
+ if qkv_delta is not None:
132
+ if qkv_delta.shape != qkv.shape:
133
+ raise RuntimeError("phase LoRA QKV delta shape mismatch")
134
+ qkv = qkv + qkv_delta
135
+ q, k, v = qkv.split([self.q_dim, self.kv_dim, self.kv_dim], dim=-1)
136
+ q = q.view(b, t, self.nh, self.hd).transpose(1, 2)
137
+ k = k.view(b, t, self.nkv, self.hd).transpose(1, 2)
138
+ v = v.view(b, t, self.nkv, self.hd).transpose(1, 2)
139
+ q = apply_rope(self.qn(q), cos, sin)
140
+ k = apply_rope(self.kn(k), cos, sin)
141
+ if self.nkv != self.nh:
142
+ repeat = self.nh // self.nkv
143
+ k = k.repeat_interleave(repeat, dim=1)
144
+ v = v.repeat_interleave(repeat, dim=1)
145
+ if attn_mask is None:
146
+ z = F.scaled_dot_product_attention(q, k, v, is_causal=True, dropout_p=0.0)
147
+ else:
148
+ z = F.scaled_dot_product_attention(q, k, v, attn_mask=attn_mask, dropout_p=0.0)
149
+ return self.out(z.transpose(1, 2).contiguous().view(b, t, self.q_dim))
150
+
151
+
152
+ class SwiGLU(nn.Module):
153
+ def __init__(self, c):
154
+ super().__init__()
155
+ self.gate_up = nn.Linear(c.hidden_size, 2 * c.intermediate_size, bias=False)
156
+ self.down = nn.Linear(c.intermediate_size, c.hidden_size, bias=False)
157
+
158
+ def forward(self, x):
159
+ gate, up = self.gate_up(x).chunk(2, dim=-1)
160
+ return self.down(F.silu(gate) * up)
161
+
162
+
163
+ class DeepDeltaResidual(nn.Module):
164
+ """Scalar Deep-Delta residual update used only in the shared recurrent body."""
165
+ def __init__(self, c):
166
+ super().__init__()
167
+ self.k_eps = c.ddl_k_eps
168
+ self.v_sigmoid_scale = c.ddl_v_sigmoid_scale
169
+ self.beta_init = c.ddl_beta_init
170
+ self.beta = nn.Linear(c.hidden_size, 1, bias=True)
171
+ self.v_proj = nn.Linear(c.hidden_size, 1, bias=True)
172
+
173
+ @torch.no_grad()
174
+ def reset_beta_bias(self):
175
+ p = min(max(self.beta_init, 0.0), 2.0) / 2.0
176
+ p = min(max(p, 1e-6), 1.0 - 1e-6)
177
+ self.beta.bias.fill_(math.log(p) - math.log(1.0 - p))
178
+
179
+ def forward(self, x, *, k_in, context):
180
+ d = k_in.size(-1)
181
+ eps_rms = (self.k_eps * self.k_eps) / d
182
+ k_rms = F.rms_norm(k_in, [d], eps=eps_rms)
183
+ scale = 1.0 / math.sqrt(d)
184
+ beta = 2.0 * torch.sigmoid(self.beta(context).float())
185
+ proj = torch.sum(k_rms * x, dim=-1, keepdim=True, dtype=torch.float32) * scale
186
+ v = torch.sigmoid(self.v_proj(x).float()) * self.v_sigmoid_scale
187
+ delta = ((beta * (v - proj)) * scale).to(dtype=x.dtype)
188
+ return x + delta * k_rms
189
+
190
+
191
+ class PlainBlock(nn.Module):
192
+ def __init__(self, c):
193
+ super().__init__()
194
+ self.attn_norm = RMSNorm(c.hidden_size, c.rms_eps)
195
+ self.attn = Attention(c)
196
+ self.ffn_norm = RMSNorm(c.hidden_size, c.rms_eps)
197
+ self.ffn = SwiGLU(c)
198
+
199
+ def forward(self, x, cos, sin, attn_mask):
200
+ x = x + self.attn(self.attn_norm(x), cos, sin, attn_mask=attn_mask)
201
+ return x + self.ffn(self.ffn_norm(x))
202
+
203
+
204
+ class ContinuousLoopConditioner(nn.Module):
205
+ def __init__(self, d):
206
+ super().__init__()
207
+ self.net = nn.Sequential(nn.Linear(8, d), nn.SiLU(), nn.Linear(d, 2 * d))
208
+
209
+ @staticmethod
210
+ def features(t, dt, device, dtype):
211
+ return torch.tensor([
212
+ t, dt,
213
+ math.sin(math.pi * t), math.cos(math.pi * t),
214
+ math.sin(2 * math.pi * t), math.cos(2 * math.pi * t),
215
+ math.log(max(dt, 1e-6)), math.log(max(1.0 - t + dt, 1e-6)),
216
+ ], device=device, dtype=dtype)
217
+
218
+ def parameters_for(self, t, dt, device, dtype):
219
+ f = self.features(t, dt, device, dtype)
220
+ scale, shift = self.net(f).chunk(2, dim=-1)
221
+ return f, scale, shift
222
+
223
+ @staticmethod
224
+ def modulate(x, scale, shift):
225
+ return x * (1.0 + 0.1 * scale) + 0.1 * shift
226
+
227
+
228
+ class PhaseLoRA(nn.Module):
229
+ def __init__(self, c):
230
+ super().__init__()
231
+ self.down = nn.Linear(c.hidden_size, c.lora_rank, bias=False)
232
+ self.up = nn.Linear(c.lora_rank, c.qkv_dim, bias=False)
233
+ self.gate = nn.Linear(8, c.lora_rank, bias=True)
234
+
235
+ def forward(self, x, phase_features):
236
+ gate = (2.0 * torch.sigmoid(self.gate(phase_features).float())).to(dtype=x.dtype)
237
+ return self.up(self.down(x) * gate)
238
+
239
+
240
+ class LoopedBlock(nn.Module):
241
+ def __init__(self, c):
242
+ super().__init__()
243
+ self.attn_norm = RMSNorm(c.hidden_size, c.rms_eps)
244
+ self.attn = Attention(c)
245
+ self.phase_lora = PhaseLoRA(c)
246
+ self.ddl_attn = DeepDeltaResidual(c)
247
+ self.ffn_norm = RMSNorm(c.hidden_size, c.rms_eps)
248
+ self.ffn = SwiGLU(c)
249
+ self.ddl_ffn = DeepDeltaResidual(c)
250
+
251
+ def forward(self, x, cos, sin, phase_scale, phase_shift, phase_features, attn_mask):
252
+ attn_context = ContinuousLoopConditioner.modulate(x, phase_scale, phase_shift)
253
+ qkv_delta = self.phase_lora(attn_context, phase_features)
254
+ x_norm = self.attn_norm(attn_context)
255
+ x = self.ddl_attn(
256
+ x,
257
+ k_in=self.attn(x_norm, cos, sin, qkv_delta=qkv_delta, attn_mask=attn_mask),
258
+ context=x_norm,
259
+ )
260
+ ffn_context = ContinuousLoopConditioner.modulate(x, phase_scale, phase_shift)
261
+ x_norm = self.ffn_norm(ffn_context)
262
+ return self.ddl_ffn(x, k_in=self.ffn(x_norm), context=x_norm)
263
+
264
+
265
+ class LoopHyperConnection(nn.Module):
266
+ """Two persistent loop lanes with per-budget read/mix/write scalars."""
267
+ def __init__(self, c):
268
+ super().__init__()
269
+ self.k = int(c.hyper_lanes)
270
+ self.max_loops = int(c.max_loops)
271
+ if self.k != 2:
272
+ raise ValueError("SparkBET is defined for two Hyper-Connection lanes")
273
+ shape = (self.max_loops, self.max_loops)
274
+ self.alpha = nn.Parameter(torch.zeros(*shape, self.k))
275
+ self.mix = nn.Parameter(torch.zeros(*shape, self.k, self.k))
276
+ self.beta = nn.Parameter(torch.zeros(*shape, self.k))
277
+ self.reset_parameters()
278
+
279
+ @torch.no_grad()
280
+ def reset_parameters(self):
281
+ self.alpha.zero_(); self.mix.zero_(); self.beta.zero_()
282
+ eye = torch.eye(self.k, device=self.mix.device, dtype=self.mix.dtype)
283
+ for budget in range(1, self.max_loops + 1):
284
+ b = budget - 1
285
+ for i in range(budget):
286
+ self.alpha[b, i, i % self.k] = 1.0
287
+ self.mix[b, i].copy_(eye)
288
+ self.beta[b, i].fill_(1.0)
289
+
290
+ def init_lanes(self, prelude_state):
291
+ return prelude_state.unsqueeze(0).expand(self.k, *prelude_state.shape)
292
+
293
+ def read(self, lanes, loops, iteration):
294
+ a = self.alpha[loops - 1, iteration].to(dtype=lanes.dtype)
295
+ return torch.einsum("k,kbtd->btd", a, lanes)
296
+
297
+ def write(self, lanes, branch_delta, loops, iteration):
298
+ m = self.mix[loops - 1, iteration].to(dtype=lanes.dtype)
299
+ b = self.beta[loops - 1, iteration].to(dtype=lanes.dtype)
300
+ mixed = torch.einsum("kj,jbtd->kbtd", m, lanes)
301
+ return mixed + b[:, None, None, None] * branch_delta.unsqueeze(0)
302
+
303
+ @staticmethod
304
+ def pool(lanes):
305
+ return lanes.mean(dim=0)
306
+
307
+
308
+ class SparkBET(nn.Module):
309
+ """Prelude -> six shared recurrent blocks -> coda, with exact loop budgets 1..8."""
310
+ def __init__(self, c=CFG):
311
+ super().__init__()
312
+ self.c = c
313
+ self.embed = nn.Embedding(c.vocab_size, c.hidden_size)
314
+ self.prelude = nn.ModuleList([PlainBlock(c) for _ in range(c.prelude_layers)])
315
+ self.body = nn.ModuleList([LoopedBlock(c) for _ in range(c.body_blocks)])
316
+ self.time_cond = ContinuousLoopConditioner(c.hidden_size)
317
+ self.loop_hyper = LoopHyperConnection(c)
318
+ self.coda = nn.ModuleList([PlainBlock(c) for _ in range(c.coda_layers)])
319
+ self.final_norm = RMSNorm(c.hidden_size, c.rms_eps)
320
+ self.apply(self._generic_init)
321
+ self._mechanism_init()
322
+
323
+ @staticmethod
324
+ def _generic_init(m):
325
+ if isinstance(m, nn.Linear):
326
+ nn.init.normal_(m.weight, 0.0, 0.02)
327
+ if m.bias is not None: nn.init.zeros_(m.bias)
328
+ elif isinstance(m, nn.Embedding):
329
+ nn.init.normal_(m.weight, 0.0, 0.02)
330
+
331
+ @torch.no_grad()
332
+ def _mechanism_init(self):
333
+ nn.init.zeros_(self.time_cond.net[-1].weight)
334
+ nn.init.zeros_(self.time_cond.net[-1].bias)
335
+ self.loop_hyper.reset_parameters()
336
+ for block in self.body:
337
+ nn.init.zeros_(block.phase_lora.up.weight)
338
+ nn.init.zeros_(block.phase_lora.gate.weight)
339
+ nn.init.zeros_(block.phase_lora.gate.bias)
340
+ for block in [*self.prelude, *self.body, *self.coda]:
341
+ nn.init.zeros_(block.attn.out.weight)
342
+ nn.init.zeros_(block.ffn.down.weight)
343
+ for block in self.body:
344
+ block.ddl_attn.reset_beta_bias(); block.ddl_ffn.reset_beta_bias()
345
+
346
+ def _run_plain(self, block, x, cos, sin, attn_mask):
347
+ if _USE_GRAD_CHECKPOINTING and self.training:
348
+ return checkpoint(block, x, cos, sin, attn_mask, use_reentrant=False)
349
+ return block(x, cos, sin, attn_mask)
350
+
351
+ def _run_looped(self, block, x, cos, sin, scale, shift, features, attn_mask):
352
+ if _USE_GRAD_CHECKPOINTING and self.training:
353
+ return checkpoint(block, x, cos, sin, scale, shift, features, attn_mask, use_reentrant=False)
354
+ return block(x, cos, sin, scale, shift, features, attn_mask)
355
+
356
+ def _readout(self, x, cos, sin, attn_mask):
357
+ h = x
358
+ for block in self.coda:
359
+ h = self._run_plain(block, h, cos, sin, attn_mask)
360
+ h = self.final_norm(h)
361
+ return F.linear(h, self.embed.weight)
362
+
363
+ @staticmethod
364
+ def _schedule(step_sizes):
365
+ if isinstance(step_sizes, int):
366
+ n = int(step_sizes)
367
+ step_sizes = uniform_steps(n)
368
+ if not step_sizes:
369
+ raise ValueError("empty refinement schedule")
370
+ values = [float(v) for v in step_sizes]
371
+ if any(v <= 0 for v in values):
372
+ raise ValueError("refinement strides must be positive")
373
+ if abs(sum(values) - 1.0) > 1e-5:
374
+ raise ValueError("refinement strides must sum to 1")
375
+ return values
376
+
377
+ def _prepare(self, input_ids, step_sizes, attention_mask=None):
378
+ steps = self._schedule(step_sizes)
379
+ loops = len(steps)
380
+ if loops > self.c.max_loops:
381
+ raise ValueError(f"loops {loops} > configured max_loops {self.c.max_loops}")
382
+ if input_ids.shape[1] > self.c.max_seq_len:
383
+ raise ValueError("context exceeds max_seq_len")
384
+ x = self.embed(input_ids)
385
+ attn_mask, pos = attention_mask_and_positions(input_ids, attention_mask)
386
+ cos, sin = rope_cos_sin(pos, self.c.head_dim, self.c.rope_theta, x.dtype)
387
+ for block in self.prelude:
388
+ x = self._run_plain(block, x, cos, sin, attn_mask)
389
+ lanes = self.loop_hyper.init_lanes(x)
390
+ shared_noise = torch.randn_like(x) if self.training and _STATE_NOISE_SIGMA > 0 else None
391
+ return steps, lanes, shared_noise, cos, sin, attn_mask
392
+
393
+ def _advance(self, lanes, steps, iteration, elapsed, shared_noise, cos, sin, attn_mask):
394
+ dt = steps[iteration]
395
+ t_mid = elapsed + 0.5 * dt
396
+ features, scale, shift = self.time_cond.parameters_for(t_mid, dt, lanes.device, lanes.dtype)
397
+ branch_input = self.loop_hyper.read(lanes, len(steps), iteration)
398
+ if shared_noise is not None:
399
+ t_end = elapsed + dt
400
+ sigma = _STATE_NOISE_SIGMA * max(0.0, 1.0 - t_end)
401
+ if sigma:
402
+ branch_input = branch_input + sigma * shared_noise
403
+ h = branch_input
404
+ for block in self.body:
405
+ h = self._run_looped(block, h, cos, sin, scale, shift, features, attn_mask)
406
+ return self.loop_hyper.write(lanes, h - branch_input, len(steps), iteration)
407
+
408
+ def _run_recurrence(self, input_ids, step_sizes, attention_mask=None, collect_states=False):
409
+ steps, lanes, noise, cos, sin, attn_mask = self._prepare(input_ids, step_sizes, attention_mask)
410
+ states = [] if collect_states else None
411
+ elapsed = 0.0
412
+ for i, dt in enumerate(steps):
413
+ lanes = self._advance(lanes, steps, i, elapsed, noise, cos, sin, attn_mask)
414
+ elapsed += dt
415
+ if collect_states: states.append(self.loop_hyper.pool(lanes))
416
+ return self.loop_hyper.pool(lanes), states, cos, sin, attn_mask
417
+
418
+ def forward(self, input_ids, step_sizes=None, attention_mask=None):
419
+ if step_sizes is None: step_sizes = uniform_steps(self.c.max_loops)
420
+ x, _, cos, sin, attn_mask = self._run_recurrence(input_ids, step_sizes, attention_mask, False)
421
+ return self._readout(x, cos, sin, attn_mask)
422
+
423
+ def forward_loop_exits(self, input_ids, step_sizes=None, attention_mask=None):
424
+ if step_sizes is None: step_sizes = uniform_steps(self.c.max_loops)
425
+ _, states, cos, sin, attn_mask = self._run_recurrence(input_ids, step_sizes, attention_mask, True)
426
+ return [self._readout(h, cos, sin, attn_mask) for h in states]
427
+
428
+ def count_params(self):
429
+ return sum(p.numel() for p in self.parameters())
430
+
431
+
432
+ # Preserve the old trainer/import name while changing the implementation.
433
+ BETFog = SparkBET
434
+
435
+
436
+ def uniform_steps(n):
437
+ n = int(n)
438
+ if not 1 <= n <= MAX_LOOPS:
439
+ raise ValueError(f"loop budget must be in [1,{MAX_LOOPS}]")
440
+ return [1.0 / n] * n
441
+
442
+
443
+ def architecture_shape_signature(model):
444
+ lines = [f"{k}:{tuple(v.shape)}:{v.dtype}" for k, v in model.state_dict().items()]
445
+ return hashlib.sha256("\n".join(lines).encode()).hexdigest()
446
+
447
+
448
+ @torch.no_grad()
449
+ def verify_architecture(model, rank=0):
450
+ expected = dict(
451
+ vocab_size=259, hidden_size=324, intermediate_size=864,
452
+ prelude_layers=1, body_blocks=6, coda_layers=1,
453
+ num_heads=6, num_kv_heads=2, head_dim=54,
454
+ lora_rank=16, hyper_lanes=2, max_seq_len=1024, max_loops=8,
455
+ rope_theta=10_000.0, rms_eps=1e-6,
456
+ ddl_beta_init=1.0, ddl_k_eps=1e-2, ddl_v_sigmoid_scale=4.0,
457
+ )
458
+ actual = asdict(model.c)
459
+ for k, v in expected.items():
460
+ if actual[k] != v:
461
+ raise AssertionError(f"Architecture drift: {k}={actual[k]} expected {v}")
462
+ if model.count_params() != EXPECTED_PARAM_COUNT:
463
+ raise AssertionError(f"Parameter drift: {model.count_params():,} != {EXPECTED_PARAM_COUNT:,}")
464
+ sig = architecture_shape_signature(model)
465
+ if sig != EXPECTED_ARCH_SHAPE_SHA256:
466
+ raise AssertionError(f"Architecture SHA drift: {sig} != {EXPECTED_ARCH_SHAPE_SHA256}")
467
+ if not torch.allclose(model.time_cond.net[-1].weight, torch.zeros_like(model.time_cond.net[-1].weight)):
468
+ raise AssertionError("time conditioner must start as identity")
469
+ for i, block in enumerate(model.body):
470
+ if not torch.allclose(block.phase_lora.up.weight, torch.zeros_like(block.phase_lora.up.weight)):
471
+ raise AssertionError(f"phase LoRA {i} up projection must start zero")
472
+ for name, ddl in (("attn", block.ddl_attn), ("ffn", block.ddl_ffn)):
473
+ beta = (2.0 * torch.sigmoid(ddl.beta.bias.float())).item()
474
+ if abs(beta - 1.0) > 1e-6:
475
+ raise AssertionError(f"body {i} {name} DDL beta init={beta}")
476
+ hc = model.loop_hyper
477
+ eye = torch.eye(hc.k, device=hc.mix.device, dtype=hc.mix.dtype)
478
+ for budget in range(1, model.c.max_loops + 1):
479
+ for i in range(budget):
480
+ expected_alpha = torch.zeros_like(hc.alpha[budget - 1, i]); expected_alpha[i % hc.k] = 1
481
+ if not torch.allclose(hc.alpha[budget - 1, i], expected_alpha): raise AssertionError("Hyper alpha init drift")
482
+ if not torch.allclose(hc.mix[budget - 1, i], eye): raise AssertionError("Hyper mix init drift")
483
+ if not torch.allclose(hc.beta[budget - 1, i], torch.ones_like(hc.beta[budget - 1, i])): raise AssertionError("Hyper beta init drift")
484
+ if rank == 0:
485
+ print("[verify] SparkBET architecture PASSED")
486
+ print(f"[verify] params: {model.count_params():,}")
487
+ print(f"[verify] physical blocks: 1 prelude + 6 shared body + 1 coda; L8 applications=50")
488
+ print(f"[verify] architecture SHA256: {sig}")
checkpoints/sparkbet9m/checkpoint-000000221000/COMPLETE ADDED
@@ -0,0 +1 @@
 
 
1
+ complete
checkpoints/sparkbet9m/checkpoint-000000221000/metadata.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "cortex-sparkbet9m-full-v1",
3
+ "step": 221000,
4
+ "lineage": "cc0c9b5484e046e3b892214e660d2b7d",
5
+ "architecture": "c85415fc50d03a23f89ccd870ebf53a3a00c5a61017c13a28f6444fdd71b72b9",
6
+ "pipeline": "0a5a3d695613ed57222d0a20ae6c0c70f27ed7c33f86c9e7d0ba4651bc87f062",
7
+ "sha256": "076b418994565028f27ed43bb6597c1e2b0f56ce1baa06827eb71921579e3fbb",
8
+ "bytes": 115622259,
9
+ "saved_at": 1789957868.5571094
10
+ }
checkpoints/sparkbet9m/checkpoint-000000221000/training.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:076b418994565028f27ed43bb6597c1e2b0f56ce1baa06827eb71921579e3fbb
3
+ size 115622259
checkpoints/sparkbet9m/checkpoint-000000222000/COMPLETE ADDED
@@ -0,0 +1 @@
 
 
1
+ complete
checkpoints/sparkbet9m/checkpoint-000000222000/metadata.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "cortex-sparkbet9m-full-v1",
3
+ "step": 222000,
4
+ "lineage": "cc0c9b5484e046e3b892214e660d2b7d",
5
+ "architecture": "c85415fc50d03a23f89ccd870ebf53a3a00c5a61017c13a28f6444fdd71b72b9",
6
+ "pipeline": "0a5a3d695613ed57222d0a20ae6c0c70f27ed7c33f86c9e7d0ba4651bc87f062",
7
+ "sha256": "661e2c3475097be19d6d7f780a10eabfad4323206f16920f3153337f8deb18dd",
8
+ "bytes": 115659315,
9
+ "saved_at": 1789958762.9045448
10
+ }
checkpoints/sparkbet9m/checkpoint-000000222000/training.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:661e2c3475097be19d6d7f780a10eabfad4323206f16920f3153337f8deb18dd
3
+ size 115659315
config.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_type": "bet",
3
+ "architectures": [
4
+ "BETForCausalLM"
5
+ ],
6
+ "step": 214215,
7
+ "vocab_size": 259,
8
+ "hidden_size": 324,
9
+ "intermediate_size": 864,
10
+ "prelude_layers": 1,
11
+ "body_blocks": 6,
12
+ "coda_layers": 1,
13
+ "num_attention_heads": 6,
14
+ "num_key_value_heads": 2,
15
+ "head_dim": 54,
16
+ "lora_rank": 16,
17
+ "hyper_lanes": 2,
18
+ "max_position_embeddings": 1024,
19
+ "max_loops": 8,
20
+ "rope_theta": 10000.0,
21
+ "rms_norm_eps": 1e-06,
22
+ "ddl_beta_init": 1.0,
23
+ "ddl_k_eps": 0.01,
24
+ "ddl_v_sigmoid_scale": 4.0,
25
+ "refinement_cycles": 8,
26
+ "use_cache": false,
27
+ "tie_word_embeddings": true,
28
+ "pad_token_id": 256,
29
+ "bos_token_id": 257,
30
+ "eos_token_id": 258,
31
+ "precision": "fp16 autocast / fp32 master",
32
+ "auto_map": {
33
+ "AutoConfig": "configuration_bet.BETConfig",
34
+ "AutoModelForCausalLM": "modeling_bet.BETForCausalLM",
35
+ "AutoTokenizer": [
36
+ "tokenization_bet.BETByteTokenizer",
37
+ null
38
+ ]
39
+ }
40
+ }
configuration_bet.py ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from transformers import PretrainedConfig
2
+
3
+
4
+ class BETConfig(PretrainedConfig):
5
+ model_type = "bet"
6
+
7
+ def __init__(
8
+ self,
9
+ vocab_size=259,
10
+ hidden_size=324,
11
+ intermediate_size=864,
12
+ prelude_layers=1,
13
+ body_blocks=6,
14
+ coda_layers=1,
15
+ num_attention_heads=6,
16
+ num_key_value_heads=2,
17
+ head_dim=54,
18
+ lora_rank=16,
19
+ hyper_lanes=2,
20
+ max_position_embeddings=1024,
21
+ max_loops=8,
22
+ rope_theta=10_000.0,
23
+ rms_norm_eps=1e-6,
24
+ ddl_beta_init=1.0,
25
+ ddl_k_eps=1e-2,
26
+ ddl_v_sigmoid_scale=4.0,
27
+ refinement_cycles=8,
28
+ use_cache=False,
29
+ tie_word_embeddings=True,
30
+ pad_token_id=256,
31
+ bos_token_id=257,
32
+ eos_token_id=258,
33
+ **kwargs,
34
+ ):
35
+ super().__init__(
36
+ pad_token_id=pad_token_id,
37
+ bos_token_id=bos_token_id,
38
+ eos_token_id=eos_token_id,
39
+ tie_word_embeddings=tie_word_embeddings,
40
+ is_encoder_decoder=False,
41
+ **kwargs,
42
+ )
43
+ self.vocab_size=int(vocab_size)
44
+ self.hidden_size=int(hidden_size)
45
+ self.intermediate_size=int(intermediate_size)
46
+ self.prelude_layers=int(prelude_layers)
47
+ self.body_blocks=int(body_blocks)
48
+ self.coda_layers=int(coda_layers)
49
+ # Common HF tooling expects num_hidden_layers even though only the body loops.
50
+ self.num_hidden_layers=int(prelude_layers+body_blocks+coda_layers)
51
+ self.num_attention_heads=int(num_attention_heads)
52
+ self.num_key_value_heads=int(num_key_value_heads)
53
+ self.head_dim=int(head_dim)
54
+ self.lora_rank=int(lora_rank)
55
+ self.hyper_lanes=int(hyper_lanes)
56
+ self.max_position_embeddings=int(max_position_embeddings)
57
+ self.max_loops=int(max_loops)
58
+ self.rope_theta=float(rope_theta)
59
+ self.rms_norm_eps=float(rms_norm_eps)
60
+ self.ddl_beta_init=float(ddl_beta_init)
61
+ self.ddl_k_eps=float(ddl_k_eps)
62
+ self.ddl_v_sigmoid_scale=float(ddl_v_sigmoid_scale)
63
+ self.refinement_cycles=int(refinement_cycles)
64
+ self.use_cache=bool(use_cache)
dataset_manifest.json ADDED
The diff for this file is too large to render. See raw diff
 
inference.py ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Small text-generation helper for an exported SparkBET repository."""
2
+ from pathlib import Path
3
+ import torch
4
+ from safetensors.torch import load_file
5
+ from bet_model import SparkBET,BETConfig,uniform_steps
6
+
7
+
8
+ class Cortex:
9
+ def __init__(self,model,device=None):
10
+ self.model=model
11
+ self.device=torch.device(device or ("cuda" if torch.cuda.is_available() else "cpu"))
12
+ self.model.to(self.device).eval()
13
+
14
+ @classmethod
15
+ def from_export(cls,folder,device=None):
16
+ folder=Path(folder);model=SparkBET(BETConfig())
17
+ state=load_file(str(folder/"model.safetensors"),device="cpu")
18
+ if state and all(k.startswith("core.") for k in state):state={k[5:]:v for k,v in state.items()}
19
+ model.load_state_dict(state,strict=True)
20
+ return cls(model,device)
21
+
22
+ def generate_ids(self,ids,max_new_tokens=128,loops=8,temperature=0.0,top_k=None):
23
+ out=list(map(int,ids))
24
+ for _ in range(int(max_new_tokens)):
25
+ current=out[-self.model.c.max_seq_len:]
26
+ x=torch.tensor([current],device=self.device,dtype=torch.long)
27
+ with torch.inference_mode(),torch.autocast(self.device.type,dtype=torch.float16,enabled=self.device.type=="cuda"):
28
+ logits=self.model(x,uniform_steps(loops))[0,-1].float()
29
+ if temperature and temperature>0:
30
+ logits=logits/float(temperature)
31
+ if top_k:
32
+ values,_=torch.topk(logits,min(int(top_k),logits.numel()));logits[logits<values[-1]]=-float("inf")
33
+ nxt=int(torch.multinomial(torch.softmax(logits,-1),1))
34
+ else:nxt=int(logits.argmax())
35
+ out.append(nxt)
36
+ if nxt==258:break
37
+ return out
38
+
39
+ def generate(self,text,max_new_tokens=128,loops=8,temperature=0.0,top_k=None):
40
+ ids=[257]+list(text.encode("utf-8"))
41
+ out=self.generate_ids(ids,max_new_tokens,loops,temperature,top_k)
42
+ body=bytes(i for i in out[1:] if 0<=i<=255)
43
+ return body.decode("utf-8",errors="replace")
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8835a993d051d3fdfad3bc0eed07849e5445851e9654c9debc4f56f97b959bb
3
+ size 37429752
modeling_bet.py ADDED
@@ -0,0 +1,76 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import torch
3
+ import torch.nn.functional as F
4
+ from transformers import PreTrainedModel, GenerationMixin
5
+ from transformers.modeling_outputs import CausalLMOutput
6
+
7
+ from .configuration_bet import BETConfig
8
+ from .bet_model import BETConfig as CoreConfig, SparkBET, uniform_steps
9
+
10
+
11
+ class BETPreTrainedModel(PreTrainedModel):
12
+ config_class=BETConfig
13
+ base_model_prefix="core"
14
+ supports_gradient_checkpointing=False
15
+ _no_split_modules=["PlainBlock","LoopedBlock"]
16
+
17
+
18
+ class BETForCausalLM(BETPreTrainedModel,GenerationMixin):
19
+ def __init__(self,config):
20
+ super().__init__(config)
21
+ core_cfg=CoreConfig(
22
+ vocab_size=config.vocab_size,
23
+ hidden_size=config.hidden_size,
24
+ intermediate_size=config.intermediate_size,
25
+ prelude_layers=config.prelude_layers,
26
+ body_blocks=config.body_blocks,
27
+ coda_layers=config.coda_layers,
28
+ num_heads=config.num_attention_heads,
29
+ num_kv_heads=config.num_key_value_heads,
30
+ head_dim=config.head_dim,
31
+ lora_rank=config.lora_rank,
32
+ hyper_lanes=config.hyper_lanes,
33
+ max_seq_len=config.max_position_embeddings,
34
+ max_loops=config.max_loops,
35
+ rope_theta=config.rope_theta,
36
+ rms_eps=config.rms_norm_eps,
37
+ ddl_beta_init=config.ddl_beta_init,
38
+ ddl_k_eps=config.ddl_k_eps,
39
+ ddl_v_sigmoid_scale=config.ddl_v_sigmoid_scale,
40
+ )
41
+ self.core=SparkBET(core_cfg)
42
+
43
+ def get_input_embeddings(self):return self.core.embed
44
+ def set_input_embeddings(self,value):self.core.embed=value
45
+ def get_output_embeddings(self):return None
46
+ def set_output_embeddings(self,value):
47
+ if value is not None:raise ValueError("SparkBET uses tied input/output embeddings")
48
+
49
+ def _cycles(self,cycles=None):
50
+ if cycles is None:
51
+ cycles=int(os.environ.get("BET_EVAL_CYCLES",self.config.refinement_cycles))
52
+ cycles=int(cycles)
53
+ if not 1<=cycles<=self.config.max_loops:
54
+ raise ValueError(f"refinement cycles must be in [1,{self.config.max_loops}]")
55
+ return cycles
56
+
57
+ def forward(
58
+ self,input_ids=None,attention_mask=None,labels=None,cycles=None,
59
+ past_key_values=None,use_cache=None,return_dict=True,**kwargs,
60
+ ):
61
+ if input_ids is None:raise ValueError("input_ids is required")
62
+ if past_key_values is not None:raise ValueError("SparkBET does not implement a KV cache")
63
+ logits=self.core(input_ids,uniform_steps(self._cycles(cycles)),attention_mask=attention_mask)
64
+ loss=None
65
+ if labels is not None:
66
+ shift_logits=logits[:,:-1].contiguous().float();shift_labels=labels[:,1:].contiguous()
67
+ loss=F.cross_entropy(shift_logits.view(-1,shift_logits.size(-1)),shift_labels.view(-1),ignore_index=-100)
68
+ if not return_dict:return tuple(v for v in (loss,logits) if v is not None)
69
+ return CausalLMOutput(loss=loss,logits=logits)
70
+
71
+ def prepare_inputs_for_generation(self,input_ids,attention_mask=None,**kwargs):
72
+ max_len=self.config.max_position_embeddings
73
+ if input_ids.shape[1]>max_len:
74
+ input_ids=input_ids[:,-max_len:]
75
+ if attention_mask is not None:attention_mask=attention_mask[:,-max_len:]
76
+ return {"input_ids":input_ids,"attention_mask":attention_mask,"use_cache":False}
records.py ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Direct UTF-8 bytes. Special IDs match BET, not the old Cortex tokenizer."""
2
+ from dataclasses import dataclass
3
+ PAD,BOS,EOS=256,257,258
4
+
5
+ class Oversize(ValueError):pass
6
+ class InvalidRecord(ValueError):pass
7
+
8
+ def ids(text):return list(text.encode('utf-8'))
9
+ def decode(tokens):return bytes(t for t in tokens if 0<=t<256).decode('utf-8',errors='replace')
10
+
11
+ def record(prefix,answer,source,limit=1024,supervise_all=False,meta=None):
12
+ p,a=ids(prefix),ids(answer)
13
+ if not a:raise InvalidRecord('Empty target: '+source)
14
+ tokens=[BOS]+p+a+[EOS]
15
+ if len(tokens)>limit+1:raise Oversize(f'{source}: {len(tokens)} IDs exceeds {limit+1}; no truncation')
16
+ weights=[0]+([1]*len(p) if supervise_all else [0]*len(p))+[1]*(len(a)+1)
17
+ return dict(ids=tokens,weights=weights,source=source,prompt_len=1+len(p),meta=meta or {})
18
+
19
+ def plain_chunks(text,source,limit=1024):
20
+ # Lossless bytes, including split UTF-8 sequences: decoder assembles the byte stream.
21
+ # No false EOS at chunk boundaries. One-token overlap predicts each byte once.
22
+ if not isinstance(text,str) or not text.strip():raise InvalidRecord('Empty/non-string text: '+source)
23
+ raw_bytes=text.encode('utf-8')
24
+ if len(raw_bytes)>1024*1024:raise Oversize('Document exceeds the 1 MiB bounded-buffer limit; rejected intact')
25
+ raw=[BOS]+list(raw_bytes)+[EOS];out=[]
26
+ for offset in range(0,len(raw)-1,limit):
27
+ chunk=raw[offset:offset+limit+1]
28
+ out.append(dict(ids=chunk,weights=[0]+[1]*(len(chunk)-1),source=source,prompt_len=1,meta={}))
29
+ return out
30
+
31
+ def validate(r,limit=1024):
32
+ assert 2<=len(r['ids'])<=limit+1
33
+ assert len(r['ids'])==len(r['weights'])
34
+ assert all(type(x)==int and 0<=x<259 for x in r['ids'])
35
+ assert all(x in (0,1) for x in r['weights']) and sum(r['weights'][1:])>0
36
+ assert r['weights'][0]==0
37
+ return r
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789334934.modal.206.0.step-000000000000 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:48d4b0c9e0a10d9de70d1b7796adbc34b57938b5dddd436b1f38d7bb714bfc66
3
+ size 135822
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789335534.modal.206.1.step-000000000171 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d9d692d8c29667f2c0042b6ca1099273f75a339e9cd700936f711dfabca78422
3
+ size 158142
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789336136.modal.206.2.step-000000000366 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40a73a39478912711ff653371947802ed165c30f04897f30afaae8c96848fab8
3
+ size 161639
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789336741.modal.206.3.step-000000000564 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:545b636e5a97e86f0804bb5ab4a4b098178e1c2e79b00433b0c34ee75faa1ca7
3
+ size 162830
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789337343.modal.206.4.step-000000000765 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:446a263b5774bb1337fb5366902e62a2f7f12c39166e31075da4b7798030ccba
3
+ size 157658
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789337947.modal.206.5.step-000000000960 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a7df742ad0a6981d0195ffae0b93cd31dfa0d22df9f4f5b5962b812b6c2f8cd1
3
+ size 157724
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789338549.modal.206.6.step-000000001156 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d93ce4f48de9ee600d1f2440fcd115d836ad7883cae14439aa2b7dcae4ed4244
3
+ size 162460
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789339151.modal.206.7.step-000000001355 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac026f039167ffa5fe49635ee70ab285d3d0d16537a5d5ac3bbbfb628372829f
3
+ size 161161
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789339756.modal.206.8.step-000000001551 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:814cfe498ab8491f860ab93ecbcb89f273fac38d7c20a89de303ac2083f496e5
3
+ size 158415
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789340360.modal.206.9.step-000000001749 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1e44cbf718e07a2d2d3aa70556e06c37eef1db4c554336ffe7142b5711ebc050
3
+ size 158044
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789340961.modal.206.10.step-000000001942 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e5a6614a645cb649c05423ad75dd894f265f84d5cf1a149c1df97ef64a766770
3
+ size 175345
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789341565.modal.206.11.step-000000002134 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:00e89ac910db88c663dce569346c16a6ca5ccf88055eb99dc5b148b7c0db9559
3
+ size 160122
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789342170.modal.206.12.step-000000002331 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0bcd7c62e5ab2fe57c4f479af442a2a770f3bbedf0155137dde5c5207a97643e
3
+ size 159525
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789342775.modal.206.13.step-000000002529 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38da17f69430ebde38d5a22548da847f5916436f362fc6e3de5bd1376d461fd2
3
+ size 162994
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789343379.modal.206.14.step-000000002730 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:65a859e91485b662b3a7626cc9a27feffc264bf99425e9150088ccf215dd4a7c
3
+ size 156370
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789343984.modal.206.15.step-000000002924 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8cb2ce2b3fcd2d8e8f5f2582f7e36d5437eafe6d6c5a9cb981cae9ba8d16e178
3
+ size 146854
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789344588.modal.206.16.step-000000003101 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9ac816ae4c45fade0d88ad7e285f8df9bb57a0a5b1ca1d948f4990b7b6e311ed
3
+ size 151434
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789345191.modal.206.17.step-000000003288 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e02af65a1383a3f137f01e91d04532598e98a3068f64a8f8a7e4c5d841838e33
3
+ size 157658
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789345794.modal.206.18.step-000000003483 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:104c93d2e63b2d6cdfed56e60f115d3a0cb63e942c05c68d255dea97578e008b
3
+ size 160770
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789346396.modal.206.19.step-000000003680 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:56279fc90af709fe03536202f965fa210ecd6f45a6b1e24b9f942268afd5a68d
3
+ size 156980
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789346999.modal.206.20.step-000000003873 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:db8225703f8e526951ce7c20ed08290d8c7935c50827d55a08f21cde267515d8
3
+ size 176699
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789347602.modal.206.21.step-000000004067 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8ee22577825c6ee932a1af6c50a6fe4455c088d9e2b60869fb538dc3f3cbb6b
3
+ size 155016
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789348204.modal.206.22.step-000000004259 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ea0f994944e23e8038acef7aff409f21a1f6bb9ec0525ea81d35c317468324a
3
+ size 158335
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789348809.modal.206.23.step-000000004455 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d746e70cc250725413495a476bdd36b77f71a17a4805c14b00587c488d6338e
3
+ size 159752
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789349415.modal.206.24.step-000000004650 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c38c43041f4a6c089a7057d4b06681960039b9144409fabec6d5271562eebca8
3
+ size 159887
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789350017.modal.206.25.step-000000004846 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d72babddad46bb93801e058dd1a45209adaf4b5af3649056393bc89775281bd
3
+ size 164137
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789350620.modal.206.26.step-000000005048 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:796c5dd39c9b54ed8a0c7a440aaef0fde0b771544f6ca47577a2bf2d4adccf89
3
+ size 160366
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789351224.modal.206.27.step-000000005247 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3d24ef9cfc1a68c7de9110d1c87aae75a336799ffc5be483456fcea04f2a2547
3
+ size 161495
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789351828.modal.206.28.step-000000005445 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c2958473bdccd11aa27b6e455465a1ab9d692657c3dc6f6e6d2d7429a4fae243
3
+ size 156817
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789352433.modal.206.29.step-000000005639 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd0a3757a843afaf08f1defb579db0d44a61513e6a39fb5348e8f6432987925f
3
+ size 158768
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789353038.modal.206.30.step-000000005834 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ebe50b499995af4d5f04cdd7b455bf65765355de8a1a6a7d26d850dcf179a5e7
3
+ size 177376
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789353643.modal.206.31.step-000000006029 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6d76a57d11f49cd0d026762e4761df29ab99bbdd6bbc4dc63bbdb45ece567990
3
+ size 159689
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789354246.modal.206.32.step-000000006227 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:78de2b1f8822d966cdc42942d6d9a8a456b50b40c45505543d82aaba5930c723
3
+ size 161588
runs/cc0c9b5484e046e3b892214e660d2b7d/events.out.tfevents.1789354851.modal.206.33.step-000000006422 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:507fff9afe7fefed5bebb50ffa25934b35d7935684391b22620e50c385a062d7
3
+ size 158801