File size: 2,970 Bytes
311660a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
{
  "name": "aether-phase1",
  "desc": "Qwen3-VL-8B base + InternViT-6B vision + MiMo-Audio + TRELLIS-SLAT 3D, 5 projectors, expanded vocab, factorized/3D-spatial M-RoPE. Phase-1 assembled+wired.",
  "total_params_B": 15.16,
  "trainable_params_B": 1.38,
  "vram_bf16_GB": 32,
  "base": {
    "repo": "SupremeD/leeworld-aether-base-pure",
    "revision": "main",
    "arch": "Qwen3-VL-8B",
    "text_hidden": 4096
  },
  "encoders": {
    "vision": {
      "repo": "OpenGVLab/InternViT-6B-448px-V2_5",
      "revision": "main",
      "hidden": 3200,
      "tokens_per_448img": 1025,
      "license": "MIT"
    },
    "audio": {
      "repo": "XiaomiMiMo/MiMo-Audio-Tokenizer",
      "code": "XiaomiMiMo/MiMo-Audio-7B-Base",
      "revision": "main",
      "hidden": 1280,
      "n_mels": 128,
      "input": "PACKED (total_frames,128)",
      "call": "encoder.encode(mel,lens,use_quantizer=False)",
      "frame_downsample": 4,
      "license": "MIT"
    },
    "geom": {
      "repo": "JeffreyXiang/TRELLIS-image-large",
      "fork": "CalebisGross/TRELLIS-AMD",
      "ckpt": "ckpts/slat_enc_swin8_B_64l8_fp16.safetensors",
      "latent": 8,
      "resolution": 64,
      "in_channels": 1024,
      "attn": "sdpa",
      "sparse_backend": "torchsparse",
      "spconv": "NOT required",
      "note": "SLatEncoder attention-only; build torchsparse from source on ROCm",
      "license": "MIT"
    }
  },
  "projectors": {
    "visual": [
      3200,
      4096
    ],
    "audio": [
      1280,
      4096
    ],
    "mv": [
      3200,
      4096
    ],
    "geom": [
      8,
      4096
    ],
    "type": "Linear-GELU-Linear",
    "cam_pose": [
      1,
      1,
      4096
    ]
  },
  "vocab": {
    "old": 151669,
    "new": 156296,
    "new_row_lo": 151669,
    "special": "21 structural + 256 timestamp + 4352 MiMo-RVQ audio"
  },
  "freeze": "backbone+encoders frozen; trainable = 5 projectors + cam_pose + NEW embed/lm_head rows [new_row_lo:] via grad-mask hook",
  "mrope": {
    "impl": "Qwen3-VL native 3-channel M-RoPE position_ids (temporal,H,W) \u2014 no kernel surgery",
    "text": "isotropic sequential (t=h=w)",
    "img": "2D grid (t const, h,w)",
    "audio": "scaled-1D time (t=i, h=w=start)",
    "3d_app": "per-view 2D grid (t=view)",
    "3d_geom": "3D-SPATIAL voxel (X,Y,Z) binned"
  },
  "shims": [
    "transformers.PreTrainedModel.all_tied_weights_keys={} (settable)",
    "flash_attn varlen SDPA shim",
    "ATTN_BACKEND=sdpa SPARSE_BACKEND=torchsparse"
  ],
  "verified": {
    "assembly": "15.16B params, 32GB VRAM, ASSEMBLY OK",
    "forward": "seq 5542, img1025+audio100+mv4100+geom300 spliced, M-RoPE max [230,230,230], logits (1,5542,156296) finite, FORWARD OK",
    "date": "2026-09-27",
    "hardware": "MI300 gfx942 ROCm (rental aefinal)"
  },
  "resume": "clone this repo; pull base+encoders per repos above; run assemble.py then forward.py; load new_modules.safetensors into the projectors+cam_pose; begin sliver alignment."
}