{ "llm_compressor_commit": "2d5242056a8a00028686a303771ffc8e2fe27d07", "compressed_tensors_commit": "e69c8dc58aa152e5f5e36e85800ec1d2e7de5271", "transformers": "5.17.0", "torch": "2.14.0+cu130", "vllm": "0.30.0", "base_architecture": "zai-org/GLM-5", "base_revision": "c183ef8c61faee82855eca1ed9bb3a9a7ce3b0b2", "fixture": "glm-moe-dsa", "variant": "source", "parameter_counts": { "architecture": "GlmMoeDsaForCausalLM", "layers_only_parameters": 12983272448, "backbone_parameters": 765051392, "mtp_parameters": 115984640, "total_parameters": 881036032, "changes": { "hidden_size": { "original": 6144, "tiny": 1536 }, "intermediate_size": { "original": 12288, "tiny": 6144 }, "moe_intermediate_size": { "original": 2048, "tiny": 1024 }, "num_hidden_layers": { "original": 78, "tiny": 4 }, "num_attention_heads": { "original": 64, "tiny": 16 }, "num_key_value_heads": { "original": 64, "tiny": 16 }, "n_routed_experts": { "original": 256, "tiny": 16 }, "num_experts_per_tok": { "original": 8, "tiny": 4 }, "max_position_embeddings": { "original": 202752, "tiny": 2048 }, "mlp_layer_types": { "original": [ "dense", "dense", "dense", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse", "sparse" ], "tiny": [ "dense", "dense", "dense", "sparse" ] }, "layer_types": { "original": [ "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention" ], "tiny": [ "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention", "deepseek_sparse_attention" ] }, "indexer_types": { "original": [ "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full", "full" ], "tiny": [ "full", "full", "full", "full" ] }, "num_mtp_layers": { "original": null, "tiny": 1 }, "mtp_layer_types": { "original": null, "tiny": [ "deepseek_sparse_attention" ] }, "mtp_mlp_layer_types": { "original": null, "tiny": [ "sparse" ] } } }, "backbone_validation": { "name": "glm-moe-dsa", "passed": true, "reloaded_perplexity": 1.4394527445768062, "model_patches": false, "backbone_parameters": 765051392, "backbone_activated_parameters_estimate": 708428288, "generation": "The capital of France is Paris is Paris is Paris is Paris is is is Paris is Paris is Paris is" }, "mtp_training": "synthetic initialized projections with trained final decoder block copied; not separately trained", "training": { "seed": 3225, "initialization": "random", "optimizer": "AdamW", "learning_rate": 0.0004, "weight_decay": 0.01, "batch_size": 2, "max_sequence_length": 160, "evaluation": "same toy corpus as training" }, "mfptq": { "passed": true, "source": "fixtures/glm-moe-dsa/source", "output": "verification/glm-moe-dsa-mfptq", "quantized_modules": 136, "mtp_quantized_modules": 56, "preserved_weights": [ "lm_head.weight", "model.embed_tokens.weight", "model.layers.3.mlp.gate.weight", "model.layers.4.mlp.gate.weight", "model.layers.4.eh_proj.weight" ], "model_instantiated": false }, "mfptq_backbone_reload": { "passed": true, "perplexity": 1.4401283097377287, "scope": "unmodified Transformers backbone reload; MTP tensors checked separately", "output": "verification/glm-moe-dsa-mfptq", "model_patches": false } }