baseten-admin commited on
Commit
985ef44
·
verified ·
1 Parent(s): 495e834

manifest Qwen/Qwen3.6-35B-A3B @ B200 (a9416d7e94093aa0)

Browse files
Qwen__Qwen3.6-35B-A3B/B200/tp2-seq262144-lora64x4/a9416d7e94093aa0/manifest.json ADDED
@@ -0,0 +1,111 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "Qwen/Qwen3.6-35B-A3B",
3
+ "gpu_type": "B200",
4
+ "tensor_parallel_size": 2,
5
+ "max_seq_length": 262144,
6
+ "enable_lora": true,
7
+ "max_lora_rank": 64,
8
+ "max_loras": 4,
9
+ "lora_target_modules": [],
10
+ "lora_target_module_preset": null,
11
+ "weights_source": null,
12
+ "cudagraph_capture_sizes": [
13
+ 1,
14
+ 2,
15
+ 4,
16
+ 8,
17
+ 16,
18
+ 32,
19
+ 64,
20
+ 128,
21
+ 192,
22
+ 256,
23
+ 384,
24
+ 512,
25
+ 640,
26
+ 768,
27
+ 896,
28
+ 1000
29
+ ],
30
+ "use_mega_aot_artifact": true,
31
+ "deep_gemm_warmup": "skip",
32
+ "enable_prefix_caching": true,
33
+ "moe_backend": null,
34
+ "enable_expert_parallel": false,
35
+ "vllm_version": "0.25.1",
36
+ "torch_version": "2.11.0+cu129",
37
+ "torch": "2.11.0+cu129",
38
+ "torch_cuda": "12.9",
39
+ "vllm": "0.25.1",
40
+ "image_tag": "baseten/baseten-weight-sync-inference:alex-tan-qwen36-35b-262k-b200-h200-83e0a41",
41
+ "caller": "github-actions:github-actions[bot]",
42
+ "model_revision": "995ad96eacd98c81ed38be0c5b274b04031597b0",
43
+ "build_id": "34672693781-1",
44
+ "opaque_sampler_payload": {
45
+ "tensor_parallel_size": 2,
46
+ "max_seq_length": 262144,
47
+ "enable_lora": true,
48
+ "max_lora_rank": 64,
49
+ "max_loras": 4,
50
+ "lora_target_modules": [],
51
+ "lora_target_module_preset": null,
52
+ "cudagraph_capture_sizes": [
53
+ 1,
54
+ 2,
55
+ 4,
56
+ 8,
57
+ 16,
58
+ 32,
59
+ 64,
60
+ 128,
61
+ 192,
62
+ 256,
63
+ 384,
64
+ 512,
65
+ 640,
66
+ 768,
67
+ 896,
68
+ 1000
69
+ ],
70
+ "use_mega_aot_artifact": true,
71
+ "deep_gemm_warmup": "skip",
72
+ "enable_prefix_caching": true,
73
+ "load_format": "fastsafetensors",
74
+ "disable_custom_all_reduce": false,
75
+ "gpu_memory_utilization": null,
76
+ "max_num_seqs": null,
77
+ "moe_backend": null,
78
+ "enable_expert_parallel": false
79
+ },
80
+ "build_profile": {
81
+ "cudagraph_capture_sizes": [
82
+ 1,
83
+ 2,
84
+ 4,
85
+ 8,
86
+ 16,
87
+ 32,
88
+ 64,
89
+ 128,
90
+ 192,
91
+ 256,
92
+ 384,
93
+ 512,
94
+ 640,
95
+ 768,
96
+ 896,
97
+ 1000
98
+ ],
99
+ "use_mega_aot_artifact": true,
100
+ "deep_gemm_warmup": "skip"
101
+ },
102
+ "ready_in_seconds_no_cache": 530.12,
103
+ "kv_cache_max_tokens": 9168871,
104
+ "kv_cache_max_concurrency": 34.976470588235294,
105
+ "kv_cache_gpu_memory_utilization": 0.92,
106
+ "cache_size_uncompressed_bytes": 868485120,
107
+ "cache_size_compressed_bytes": 38591968,
108
+ "compression": "zstd -9 -T0 (multithreaded)",
109
+ "compress_time_seconds": 0.86,
110
+ "upload_time_seconds": 5.08
111
+ }