{ "name": "mini-beatrix-2.5s", "d_model": 1024, "n_layers": 20, "n_heads": 16, "context": 4096, "vocab_size": 256, "tokenizer": "byte-trigram", "hub_layers": [ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19 ], "hub_K": 64, "hub_D": 128, "tau": 0.1, "bank_experts": 3, "bank_ff": 1024, "head_K": 256, "head_D": 256, "gate_init": -3.0, "tie_embeddings": false, "hub_chunk": 256, "hub_const": 4, "hub_ckpt": 0, "model_type": "mini-beatrix", "architectures": [ "MiniBeatrixForCausalLM" ], "auto_map": { "AutoConfig": "modeling_minibeatrix.MiniBeatrixConfig", "AutoModelForCausalLM": "modeling_minibeatrix.MiniBeatrixForCausalLM" }, "torch_dtype": "float32", "use_cache": false, "trunk": { "craft": "mini-beatrix-2s", "step": 61422, "base_model_id": "alephllm/mini-beatrix-2s@step61422", "note": "the core weights are bit-identical to AbstractPhil/mini-beatrix-2s; 2.5s adds the arm library" }, "default_arm": null }