baseten-admin commited on
Commit
ddc7626
·
verified ·
1 Parent(s): d7a4e4b

manifest moonshotai/Kimi-K2.6 @ B300 (a07bdfb90508b69a)

Browse files
moonshotai__Kimi-K2.6/B300/tp4-seq262144-lora64x4/a07bdfb90508b69a/manifest.json ADDED
@@ -0,0 +1,106 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "moonshotai/Kimi-K2.6",
3
+ "gpu_type": "B300",
4
+ "tensor_parallel_size": 4,
5
+ "max_seq_length": 262144,
6
+ "enable_lora": true,
7
+ "max_lora_rank": 64,
8
+ "max_loras": 4,
9
+ "lora_target_modules": [],
10
+ "lora_target_module_preset": null,
11
+ "weights_source": null,
12
+ "cudagraph_capture_sizes": [
13
+ 1,
14
+ 2,
15
+ 4,
16
+ 8,
17
+ 16,
18
+ 32,
19
+ 64,
20
+ 128,
21
+ 192,
22
+ 256,
23
+ 384,
24
+ 512,
25
+ 640,
26
+ 768,
27
+ 896,
28
+ 1000
29
+ ],
30
+ "use_mega_aot_artifact": true,
31
+ "deep_gemm_warmup": "skip",
32
+ "enable_prefix_caching": true,
33
+ "vllm_version": "0.25.1",
34
+ "torch_version": "2.11.0+cu130",
35
+ "torch": "2.11.0+cu130",
36
+ "torch_cuda": "13.0",
37
+ "vllm": "0.25.1",
38
+ "image_tag": "baseten/baseten-weight-sync-inference:will-cu130-sampler-image-3fb7b93-cu130",
39
+ "caller": "github-actions:William-Gao1",
40
+ "model_revision": "7eb5002f6aadc958aed6a9177b7ed26bb94011bb",
41
+ "build_id": "29977817476-1",
42
+ "opaque_sampler_payload": {
43
+ "tensor_parallel_size": 4,
44
+ "max_seq_length": 262144,
45
+ "enable_lora": true,
46
+ "max_lora_rank": 64,
47
+ "max_loras": 4,
48
+ "lora_target_modules": [],
49
+ "lora_target_module_preset": null,
50
+ "cudagraph_capture_sizes": [
51
+ 1,
52
+ 2,
53
+ 4,
54
+ 8,
55
+ 16,
56
+ 32,
57
+ 64,
58
+ 128,
59
+ 192,
60
+ 256,
61
+ 384,
62
+ 512,
63
+ 640,
64
+ 768,
65
+ 896,
66
+ 1000
67
+ ],
68
+ "use_mega_aot_artifact": true,
69
+ "deep_gemm_warmup": "skip",
70
+ "enable_prefix_caching": true,
71
+ "load_format": "fastsafetensors",
72
+ "disable_custom_all_reduce": false,
73
+ "gpu_memory_utilization": null
74
+ },
75
+ "build_profile": {
76
+ "cudagraph_capture_sizes": [
77
+ 1,
78
+ 2,
79
+ 4,
80
+ 8,
81
+ 16,
82
+ 32,
83
+ 64,
84
+ 128,
85
+ 192,
86
+ 256,
87
+ 384,
88
+ 512,
89
+ 640,
90
+ 768,
91
+ 896,
92
+ 1000
93
+ ],
94
+ "use_mega_aot_artifact": true,
95
+ "deep_gemm_warmup": "skip"
96
+ },
97
+ "ready_in_seconds_no_cache": 764.19,
98
+ "kv_cache_max_tokens": 1470368,
99
+ "kv_cache_max_concurrency": 5.6090087890625,
100
+ "kv_cache_gpu_memory_utilization": 0.92,
101
+ "cache_size_uncompressed_bytes": 283074560,
102
+ "cache_size_compressed_bytes": 7372365,
103
+ "compression": "zstd -9 -T0 (multithreaded)",
104
+ "compress_time_seconds": 0.25,
105
+ "upload_time_seconds": 7.05
106
+ }