Download MANIFEST.json from SupremeD/aether-phase1: direct link, hf CLI and curl.
- Browser
- Download file 2.97 kB
-
https://huggingface.co/SupremeD/aether-phase1/resolve/main/MANIFEST.json
- Command line
-
hf download hf://SupremeD/aether-phase1/MANIFEST.json
-
curl -L -o MANIFEST.json https://huggingface.co/SupremeD/aether-phase1/resolve/main/MANIFEST.json
2.97 kB
| { | |
| "name": "aether-phase1", | |
| "desc": "Qwen3-VL-8B base + InternViT-6B vision + MiMo-Audio + TRELLIS-SLAT 3D, 5 projectors, expanded vocab, factorized/3D-spatial M-RoPE. Phase-1 assembled+wired.", | |
| "total_params_B": 15.16, | |
| "trainable_params_B": 1.38, | |
| "vram_bf16_GB": 32, | |
| "base": { | |
| "repo": "SupremeD/leeworld-aether-base-pure", | |
| "revision": "main", | |
| "arch": "Qwen3-VL-8B", | |
| "text_hidden": 4096 | |
| }, | |
| "encoders": { | |
| "vision": { | |
| "repo": "OpenGVLab/InternViT-6B-448px-V2_5", | |
| "revision": "main", | |
| "hidden": 3200, | |
| "tokens_per_448img": 1025, | |
| "license": "MIT" | |
| }, | |
| "audio": { | |
| "repo": "XiaomiMiMo/MiMo-Audio-Tokenizer", | |
| "code": "XiaomiMiMo/MiMo-Audio-7B-Base", | |
| "revision": "main", | |
| "hidden": 1280, | |
| "n_mels": 128, | |
| "input": "PACKED (total_frames,128)", | |
| "call": "encoder.encode(mel,lens,use_quantizer=False)", | |
| "frame_downsample": 4, | |
| "license": "MIT" | |
| }, | |
| "geom": { | |
| "repo": "JeffreyXiang/TRELLIS-image-large", | |
| "fork": "CalebisGross/TRELLIS-AMD", | |
| "ckpt": "ckpts/slat_enc_swin8_B_64l8_fp16.safetensors", | |
| "latent": 8, | |
| "resolution": 64, | |
| "in_channels": 1024, | |
| "attn": "sdpa", | |
| "sparse_backend": "torchsparse", | |
| "spconv": "NOT required", | |
| "note": "SLatEncoder attention-only; build torchsparse from source on ROCm", | |
| "license": "MIT" | |
| } | |
| }, | |
| "projectors": { | |
| "visual": [ | |
| 3200, | |
| 4096 | |
| ], | |
| "audio": [ | |
| 1280, | |
| 4096 | |
| ], | |
| "mv": [ | |
| 3200, | |
| 4096 | |
| ], | |
| "geom": [ | |
| 8, | |
| 4096 | |
| ], | |
| "type": "Linear-GELU-Linear", | |
| "cam_pose": [ | |
| 1, | |
| 1, | |
| 4096 | |
| ] | |
| }, | |
| "vocab": { | |
| "old": 151669, | |
| "new": 156296, | |
| "new_row_lo": 151669, | |
| "special": "21 structural + 256 timestamp + 4352 MiMo-RVQ audio" | |
| }, | |
| "freeze": "backbone+encoders frozen; trainable = 5 projectors + cam_pose + NEW embed/lm_head rows [new_row_lo:] via grad-mask hook", | |
| "mrope": { | |
| "impl": "Qwen3-VL native 3-channel M-RoPE position_ids (temporal,H,W) \u2014 no kernel surgery", | |
| "text": "isotropic sequential (t=h=w)", | |
| "img": "2D grid (t const, h,w)", | |
| "audio": "scaled-1D time (t=i, h=w=start)", | |
| "3d_app": "per-view 2D grid (t=view)", | |
| "3d_geom": "3D-SPATIAL voxel (X,Y,Z) binned" | |
| }, | |
| "shims": [ | |
| "transformers.PreTrainedModel.all_tied_weights_keys={} (settable)", | |
| "flash_attn varlen SDPA shim", | |
| "ATTN_BACKEND=sdpa SPARSE_BACKEND=torchsparse" | |
| ], | |
| "verified": { | |
| "assembly": "15.16B params, 32GB VRAM, ASSEMBLY OK", | |
| "forward": "seq 5542, img1025+audio100+mv4100+geom300 spliced, M-RoPE max [230,230,230], logits (1,5542,156296) finite, FORWARD OK", | |
| "date": "2026-09-27", | |
| "hardware": "MI300 gfx942 ROCm (rental aefinal)" | |
| }, | |
| "resume": "clone this repo; pull base+encoders per repos above; run assemble.py then forward.py; load new_modules.safetensors into the projectors+cam_pose; begin sliver alignment." | |
| } |