baseten-admin commited on
Commit
bedc151
·
verified ·
1 Parent(s): e093a8f

manifest nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16 @ B200 (07c59cc4d89b61bb)

Browse files
nvidia__NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16/B200/tp8-seq131072-lora64x1/07c59cc4d89b61bb/manifest.json ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
3
+ "gpu_type": "B200",
4
+ "tensor_parallel_size": 8,
5
+ "max_seq_length": 131072,
6
+ "enable_lora": true,
7
+ "max_lora_rank": 64,
8
+ "max_loras": 1,
9
+ "cudagraph_capture_sizes": [
10
+ 1,
11
+ 2,
12
+ 4,
13
+ 8,
14
+ 16,
15
+ 32,
16
+ 64,
17
+ 128,
18
+ 192,
19
+ 256,
20
+ 384,
21
+ 512
22
+ ],
23
+ "use_mega_aot_artifact": true,
24
+ "deep_gemm_warmup": "skip",
25
+ "enable_prefix_caching": true,
26
+ "vllm_version": "0.22.0",
27
+ "torch_version": "2.11.0+cu129",
28
+ "torch": "2.11.0+cu129",
29
+ "torch_cuda": "12.9",
30
+ "vllm": "0.22.0",
31
+ "image_tag": "baseten/baseten-weight-sync-inference:main-15e6be27",
32
+ "caller": "github-actions:github-actions[bot]",
33
+ "model_revision": "624ba927cfbef0427354998700de3d51173c8c04",
34
+ "build_id": "28193124938-1",
35
+ "opaque_sampler_payload": {
36
+ "tensor_parallel_size": 8,
37
+ "max_seq_length": 131072,
38
+ "enable_lora": true,
39
+ "max_lora_rank": 64,
40
+ "max_loras": 1,
41
+ "cudagraph_capture_sizes": [
42
+ 1,
43
+ 2,
44
+ 4,
45
+ 8,
46
+ 16,
47
+ 32,
48
+ 64,
49
+ 128,
50
+ 192,
51
+ 256,
52
+ 384,
53
+ 512
54
+ ],
55
+ "use_mega_aot_artifact": true,
56
+ "deep_gemm_warmup": "skip",
57
+ "enable_prefix_caching": true,
58
+ "load_format": "fastsafetensors",
59
+ "disable_custom_all_reduce": true
60
+ },
61
+ "build_profile": {
62
+ "cudagraph_capture_sizes": [
63
+ 1,
64
+ 2,
65
+ 4,
66
+ 8,
67
+ 16,
68
+ 32,
69
+ 64,
70
+ 128,
71
+ 192,
72
+ 256,
73
+ 384,
74
+ 512
75
+ ],
76
+ "use_mega_aot_artifact": true,
77
+ "deep_gemm_warmup": "skip"
78
+ },
79
+ "ready_in_seconds_no_cache": 522.1,
80
+ "kv_cache_max_tokens": 366142,
81
+ "kv_cache_max_concurrency": 2.7934426229508196,
82
+ "kv_cache_gpu_memory_utilization": 0.92,
83
+ "cache_size_uncompressed_bytes": 1087150080,
84
+ "cache_size_compressed_bytes": 74665461,
85
+ "compression": "zstd -9 -T0 (multithreaded)",
86
+ "compress_time_seconds": 2.15,
87
+ "upload_time_seconds": 2.19
88
+ }