StephaneGSL/kimi-k3-analysis / data /architecture.json
StephaneGSL's picture
download
raw
1.37 kB
{
"$schema": "../schemas/architecture.schema.json",
"model": "moonshotai/Kimi-K3",
"snapshotDate": "2026-07-27",
"architecture": "Mixture-of-Experts",
"parameters": {
"total": 2800000000000,
"activePerToken": 104000000000,
"releasedWeightBytes": 1560936091448,
"weightShards": 96
},
"backbone": {
"layers": 93,
"hiddenDimension": 7168,
"attentionHeads": 96,
"attentionComposition": {
"kdaLayers": 69,
"gatedMlaLayers": 24,
"pattern": "3 KDA + 1 Gated MLA"
},
"attentionResidualBlockSize": 12,
"activation": "SiTU-GLU",
"contextLength": 1048576,
"vocabularySize": 163840
},
"moe": {
"routedExperts": 896,
"selectedExpertsPerToken": 16,
"sharedExperts": 2,
"latentDimension": 3584,
"expertHiddenDimension": 3072,
"routerActivation": "sigmoid",
"loadBalancing": "Quantile Balancing"
},
"vision": {
"encoder": "MoonViT-V2",
"parameters": 401000000,
"layers": 27,
"patchSize": 14,
"attentionHeads": 12,
"pixelShuffleReduction": 4
},
"quantization": {
"expertWeights": "MXFP4",
"expertActivations": "MXFP8",
"method": "quantization-aware post-training"
},
"publicImplementation": {
"kind": "Transformers reference implementation",
"trainingCodeComplete": false,
"productionKernelsIncluded": false
}
}

Xet Storage Details

Size:
1.37 kB
·
Xet hash:
08e7d5aa4a58a30a6504e09966c46aa621c755c8133018b0ea12fec3a8da3d99

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.