baseten-admin commited on
Commit
1dab01d
·
verified ·
1 Parent(s): 313c3f5

manifest zai-org/GLM-5.3-Flash @ B200 (85fecedca608f809)

Browse files
zai-org__GLM-5.3-Flash/B200/tp4-seq262144-lora128x1/85fecedca608f809/manifest.json ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "zai-org/GLM-5.3-Flash",
3
+ "gpu_type": "B200",
4
+ "tensor_parallel_size": 4,
5
+ "max_seq_length": 262144,
6
+ "enable_lora": true,
7
+ "max_lora_rank": 128,
8
+ "max_loras": 1,
9
+ "lora_target_modules": [
10
+ "q_a_proj",
11
+ "kv_a_proj_with_mqa",
12
+ "q_b_proj",
13
+ "q_proj",
14
+ "k_proj",
15
+ "v_proj",
16
+ "b_proj",
17
+ "f_a_proj",
18
+ "f_b_proj",
19
+ "g_a_proj",
20
+ "g_b_proj",
21
+ "o_proj",
22
+ "gate_proj",
23
+ "up_proj",
24
+ "down_proj",
25
+ "lm_head"
26
+ ],
27
+ "lora_target_module_preset": null,
28
+ "weights_source": null,
29
+ "cudagraph_capture_sizes": [
30
+ 1,
31
+ 2,
32
+ 4,
33
+ 8,
34
+ 16,
35
+ 32,
36
+ 64,
37
+ 128,
38
+ 192,
39
+ 256,
40
+ 384,
41
+ 512,
42
+ 640,
43
+ 768,
44
+ 896,
45
+ 1000
46
+ ],
47
+ "use_mega_aot_artifact": true,
48
+ "deep_gemm_warmup": "skip",
49
+ "enable_prefix_caching": true,
50
+ "moe_backend": null,
51
+ "enable_expert_parallel": false,
52
+ "vllm_version": "0.1.dev20051+g487ecf187",
53
+ "torch_version": "2.13.0+cu130",
54
+ "torch": "2.13.0+cu130",
55
+ "torch_cuda": "13.0",
56
+ "vllm": "0.1.dev20051+g487ecf187",
57
+ "image_tag": "baseten/baseten-weight-sync-inference:jerry-glm53-flash-h200-sampler-b4c1939-glm53-flash",
58
+ "caller": "github-actions:github-actions[bot]",
59
+ "model_revision": "eb9eb208eb0d988989d07a6a12d0fdeb5f52574a",
60
+ "build_id": "34430498285-1",
61
+ "opaque_sampler_payload": {
62
+ "tensor_parallel_size": 4,
63
+ "max_seq_length": 262144,
64
+ "enable_lora": true,
65
+ "max_lora_rank": 128,
66
+ "max_loras": 1,
67
+ "lora_target_modules": [
68
+ "q_a_proj",
69
+ "kv_a_proj_with_mqa",
70
+ "q_b_proj",
71
+ "q_proj",
72
+ "k_proj",
73
+ "v_proj",
74
+ "b_proj",
75
+ "f_a_proj",
76
+ "f_b_proj",
77
+ "g_a_proj",
78
+ "g_b_proj",
79
+ "o_proj",
80
+ "gate_proj",
81
+ "up_proj",
82
+ "down_proj",
83
+ "lm_head"
84
+ ],
85
+ "lora_target_module_preset": null,
86
+ "cudagraph_capture_sizes": [
87
+ 1,
88
+ 2,
89
+ 4,
90
+ 8,
91
+ 16,
92
+ 32,
93
+ 64,
94
+ 128,
95
+ 192,
96
+ 256,
97
+ 384,
98
+ 512,
99
+ 640,
100
+ 768,
101
+ 896,
102
+ 1000
103
+ ],
104
+ "use_mega_aot_artifact": true,
105
+ "deep_gemm_warmup": "skip",
106
+ "enable_prefix_caching": true,
107
+ "load_format": "auto",
108
+ "disable_custom_all_reduce": false,
109
+ "gpu_memory_utilization": 0.9,
110
+ "max_num_seqs": null,
111
+ "moe_backend": null,
112
+ "enable_expert_parallel": false
113
+ },
114
+ "build_profile": {
115
+ "cudagraph_capture_sizes": [
116
+ 1,
117
+ 2,
118
+ 4,
119
+ 8,
120
+ 16,
121
+ 32,
122
+ 64,
123
+ 128,
124
+ 192,
125
+ 256,
126
+ 384,
127
+ 512,
128
+ 640,
129
+ 768,
130
+ 896,
131
+ 1000
132
+ ],
133
+ "use_mega_aot_artifact": true,
134
+ "deep_gemm_warmup": "skip"
135
+ },
136
+ "ready_in_seconds_no_cache": 1776.5,
137
+ "kv_cache_max_tokens": 12121135,
138
+ "kv_cache_max_concurrency": 46.238461538461536,
139
+ "kv_cache_gpu_memory_utilization": 0.9,
140
+ "cache_size_uncompressed_bytes": 4188160,
141
+ "cache_size_compressed_bytes": 198983,
142
+ "compression": "zstd -9 -T0 (multithreaded)",
143
+ "compress_time_seconds": 0.05,
144
+ "upload_time_seconds": 3.62
145
+ }