baseten-admin commited on
Commit
abe5d51
·
verified ·
1 Parent(s): f038cdd

manifest Qwen/Qwen3-30B-A3B-Instruct-2507 @ B300 (4a85dbcdb0b305ea)

Browse files
Qwen__Qwen3-30B-A3B-Instruct-2507/B300/tp1-seq131072-lora64x4/4a85dbcdb0b305ea/manifest.json ADDED
@@ -0,0 +1,106 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "Qwen/Qwen3-30B-A3B-Instruct-2507",
3
+ "gpu_type": "B300",
4
+ "tensor_parallel_size": 1,
5
+ "max_seq_length": 131072,
6
+ "enable_lora": true,
7
+ "max_lora_rank": 64,
8
+ "max_loras": 4,
9
+ "lora_target_modules": [],
10
+ "lora_target_module_preset": null,
11
+ "weights_source": null,
12
+ "cudagraph_capture_sizes": [
13
+ 1,
14
+ 2,
15
+ 4,
16
+ 8,
17
+ 16,
18
+ 32,
19
+ 64,
20
+ 128,
21
+ 192,
22
+ 256,
23
+ 384,
24
+ 512,
25
+ 640,
26
+ 768,
27
+ 896,
28
+ 1000
29
+ ],
30
+ "use_mega_aot_artifact": true,
31
+ "deep_gemm_warmup": "skip",
32
+ "enable_prefix_caching": true,
33
+ "vllm_version": "0.25.1",
34
+ "torch_version": "2.11.0+cu129",
35
+ "torch": "2.11.0+cu129",
36
+ "torch_cuda": "12.9",
37
+ "vllm": "0.25.1",
38
+ "image_tag": "baseten/baseten-weight-sync-inference:main-68cee18",
39
+ "caller": "github-actions:William-Gao1",
40
+ "model_revision": "0d7cf23991f47feeb3a57ecb4c9cee8ea4a17bfe",
41
+ "build_id": "29973684518-1",
42
+ "opaque_sampler_payload": {
43
+ "tensor_parallel_size": 1,
44
+ "max_seq_length": 131072,
45
+ "enable_lora": true,
46
+ "max_lora_rank": 64,
47
+ "max_loras": 4,
48
+ "lora_target_modules": [],
49
+ "lora_target_module_preset": null,
50
+ "cudagraph_capture_sizes": [
51
+ 1,
52
+ 2,
53
+ 4,
54
+ 8,
55
+ 16,
56
+ 32,
57
+ 64,
58
+ 128,
59
+ 192,
60
+ 256,
61
+ 384,
62
+ 512,
63
+ 640,
64
+ 768,
65
+ 896,
66
+ 1000
67
+ ],
68
+ "use_mega_aot_artifact": true,
69
+ "deep_gemm_warmup": "skip",
70
+ "enable_prefix_caching": true,
71
+ "load_format": "fastsafetensors",
72
+ "disable_custom_all_reduce": false,
73
+ "gpu_memory_utilization": null
74
+ },
75
+ "build_profile": {
76
+ "cudagraph_capture_sizes": [
77
+ 1,
78
+ 2,
79
+ 4,
80
+ 8,
81
+ 16,
82
+ 32,
83
+ 64,
84
+ 128,
85
+ 192,
86
+ 256,
87
+ 384,
88
+ 512,
89
+ 640,
90
+ 768,
91
+ 896,
92
+ 1000
93
+ ],
94
+ "use_mega_aot_artifact": true,
95
+ "deep_gemm_warmup": "skip"
96
+ },
97
+ "ready_in_seconds_no_cache": 320.09,
98
+ "kv_cache_max_tokens": 1752032,
99
+ "kv_cache_max_concurrency": 13.366943359375,
100
+ "kv_cache_gpu_memory_utilization": 0.92,
101
+ "cache_size_uncompressed_bytes": 74813440,
102
+ "cache_size_compressed_bytes": 4113709,
103
+ "compression": "zstd -9 -T0 (multithreaded)",
104
+ "compress_time_seconds": 0.24,
105
+ "upload_time_seconds": 6.81
106
+ }