From e4d186b8329adcbe02f2d1e54dab6cb4cafada67 Mon Sep 17 00:00:00 2001 From: Yifan Qiao Date: Mon, 28 Sep 2026 10:17:00 -0500 Subject: [PATCH 1/2] perf(b300): update vLLM AgentX to DSpark6 Move dsv4-fp4-b300-vllm-agentic-mtp to DeepSeek-V4-Pro-0813 with DSpark (six draft tokens) on vllm-openai nightly-dev-x86_64-cu13.0.1-20ebb28, with a sampled concurrency grid: TP8 c1-c16, TP4 c8, and DEP8 c64-c288. The config now runs from its srt-slurm recipe, so the former script changes live in the recipe's variants: max-num-seqs at CONC (DEP8 at 2x CONC across ranks), decode graphs for both draft and verification shapes with TP piecewise sizes, DEP8 KV cache byte pins and an eight-aligned token budget, per-rank eager SimpleCPUOffload, NUMA binding on all 192 CPUs, and a two-hour engine readiness window. The launcher reads the 0813 checkpoint from node-local NVMe for vLLM. --- .../dsv4/vllm/b300-fp4-mtp/agentic.yaml | 264 +++++++----------- inferencex-e2e/configs/nvidia-master.yaml | 20 +- inferencex-e2e/perf-changelog.yaml | 8 + inferencex-e2e/runners/launch_b300-dsxe.sh | 5 +- 4 files changed, 119 insertions(+), 178 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml index 800c798e7b..ce329a2947 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml @@ -1,12 +1,12 @@ -# DeepSeek-V4-Pro AgentX on B300 with vLLM native MTP (three draft tokens). -# TP8 and TP4 c8 are GPU-resident; TP4 c16, DEP4 and DEP8 offload KV to host -# DRAM through SimpleCPUOffloadConnector. +# DeepSeek-V4-Pro-0813 AgentX on B300 with vLLM DSpark (six draft tokens). +# TP points are GPU-resident; DEP8 offloads KV to host DRAM through +# SimpleCPUOffloadConnector. base: schema: 2 name: dsv4-fp4-b300-vllm-agentic model: - path: hf:deepseek-ai/DeepSeek-V4-Pro - container: vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-426e59f + path: hf:deepseek-ai/DeepSeek-V4-Pro-0813 + container: vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-20ebb28 precision: fp4 resources: gpu_type: b300 @@ -24,22 +24,26 @@ base: # The engine may take the full VLLM_ENGINE_READY_TIMEOUT_S to become ready. health_check: interval_seconds: 10 - max_attempts: 360 + max_attempts: 720 + # --numa-bind pins each rank to its GPU's socket, which needs every host CPU. + sbatch_directives: + cpus-per-task: "192" roles: agg: nodes: 1 workers: 1 args: - served-model-name: deepseek-ai/DeepSeek-V4-Pro + served-model-name: deepseek-ai/DeepSeek-V4-Pro-0813 + numa-bind: true trust-remote-code: true no-enable-flashinfer-autotune: true no-disable-hybrid-kv-cache-manager: true kv-cache-dtype: fp8 block-size: 256 max-model-len: 1048576 - attention-config: '{"use_fp4_indexer_cache":true,"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true}' + attention-config: '{"indexer_kv_dtype":"mxfp4","backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true}' # Throughput runs switch to synthetic rejection at the golden acceptance length. - speculative-config: '{"method":"mtp","num_speculative_tokens":3}' + speculative-config: '{"method":"dspark","num_speculative_tokens":6,"draft_sample_method":"probabilistic"}' disable-uvicorn-access-log: true tokenizer-mode: deepseek_v4 tool-call-parser: deepseek_v4 @@ -47,7 +51,7 @@ base: reasoning-parser: deepseek_v4 env: VLLM_USE_V2_MODEL_RUNNER: '1' - VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_ENGINE_READY_TIMEOUT_S: '7200' VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '32768' VLLM_DSV4_MEGA_FP8_COMBINE: '1' NCCL_NVLS_ENABLE: '1' @@ -59,17 +63,20 @@ base: type: custom command: bash /infmax-workspace/benchmarks/srt_agentic.sh env: - MODEL: deepseek-ai/DeepSeek-V4-Pro + MODEL: deepseek-ai/DeepSeek-V4-Pro-0813 AIPERF_REQUIRED_SERVER_METRIC_PREFIX: 'vllm:' -# One variant per point. Decode graphs capture every batch of up to -# max-num-seqs sequences in tokens (1 target + 3 drafts each). TP points admit -# 2x CONC with the FlashInfer all-reduce; DEP points run one data-parallel rank -# per GPU behind a consistent-hash vLLM Router (turns of one conversation share -# a rank), admit 2x CONC across the ranks with MegaMoE experts, and keep eager -# offload so block hashes agree across ranks. DEP8 takes a larger prefill -# budget and more memory headroom. DRAM points split TOTAL_CPU_DRAM_GB GB -# across the ranks. +# One variant per point. Target verification uses 7 tokens per sequence and +# DSpark drafting 6, so decode graphs capture both shapes up to max-num-seqs. +# TP points admit CONC sequences with the FlashInfer all-reduce and add +# piecewise shapes for mixed prefill/decode batches. DEP8 runs one +# data-parallel rank per GPU behind a consistent-hash vLLM Router (turns of one +# conversation share a rank), admits 2x CONC across the ranks with MegaMoE +# experts, keeps eager offload so block hashes agree across ranks, pins KV +# cache bytes per concurrency tier, and sizes its token budget to the 8192-token +# prefill budget plus one verification step per sequence, rounded up to a +# multiple of eight. DRAM points split TOTAL_CPU_DRAM_GB GB across the ranks. + override_tp8_c1: roles: agg: @@ -79,8 +86,8 @@ override_tp8_c1: data-parallel-size: 1 disable-custom-all-reduce: true gpu-memory-utilization: 0.95 - max-num-seqs: 2 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8],"mode":0}' + max-num-seqs: 1 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,100,200,300,400,500]}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' @@ -89,7 +96,7 @@ override_tp8_c1: CONC: '1' KV_OFFLOADING: none -override_tp8_c4: +override_tp8_c2: roles: agg: gpus: 8 @@ -98,90 +105,71 @@ override_tp8_c4: data-parallel-size: 1 disable-custom-all-reduce: true gpu-memory-utilization: 0.95 - max-num-seqs: 8 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32],"mode":0}' - env: - VLLM_ALLREDUCE_USE_FLASHINFER: '1' - VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' - benchmark: - env: - CONC: '4' - KV_OFFLOADING: none - -override_tp4_c1: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - data-parallel-size: 1 - disable-custom-all-reduce: true - gpu-memory-utilization: 0.95 max-num-seqs: 2 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8],"mode":0}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,100,200,300,400,500]}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' benchmark: env: - CONC: '1' + CONC: '2' KV_OFFLOADING: none -override_tp4_c2: +override_tp8_c4: roles: agg: - gpus: 4 + gpus: 8 args: - tensor-parallel-size: 4 + tensor-parallel-size: 8 data-parallel-size: 1 disable-custom-all-reduce: true gpu-memory-utilization: 0.95 max-num-seqs: 4 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16],"mode":0}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,100,200,300,400,500]}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' benchmark: env: - CONC: '2' + CONC: '4' KV_OFFLOADING: none -override_tp4_c4: +override_tp8_c8: roles: agg: - gpus: 4 + gpus: 8 args: - tensor-parallel-size: 4 + tensor-parallel-size: 8 data-parallel-size: 1 disable-custom-all-reduce: true gpu-memory-utilization: 0.95 max-num-seqs: 8 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32],"mode":0}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,56,100,200,300,400,500]}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' benchmark: env: - CONC: '4' + CONC: '8' KV_OFFLOADING: none -override_tp4_c6: +override_tp8_c16: roles: agg: - gpus: 4 + gpus: 8 args: - tensor-parallel-size: 4 + tensor-parallel-size: 8 data-parallel-size: 1 disable-custom-all-reduce: true gpu-memory-utilization: 0.95 - max-num-seqs: 12 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48],"mode":0}' + max-num-seqs: 16 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,100,105,112,200,300,400,500]}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' benchmark: env: - CONC: '6' + CONC: '16' KV_OFFLOADING: none override_tp4_c8: @@ -193,8 +181,8 @@ override_tp4_c8: data-parallel-size: 1 disable-custom-all-reduce: true gpu-memory-utilization: 0.95 - max-num-seqs: 16 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}' + max-num-seqs: 8 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,56,100,200,300,400,500]}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' @@ -203,29 +191,7 @@ override_tp4_c8: CONC: '8' KV_OFFLOADING: none -override_tp4_c16: - roles: - agg: - gpus: 4 - args: - tensor-parallel-size: 4 - data-parallel-size: 1 - disable-custom-all-reduce: true - gpu-memory-utilization: 0.95 - max-num-seqs: 32 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356000000000,"enable_cross_layers_blocks":"true","lazy_offload":true}}' - env: - VLLM_ALLREDUCE_USE_FLASHINFER: '1' - VLLM_FLASHINFER_ALLREDUCE_BACKEND: 'auto' - PYTHONHASHSEED: '42' - benchmark: - env: - CONC: '16' - KV_OFFLOADING: dram - TOTAL_CPU_DRAM_GB: '1424' - -override_dep4_c48: +override_dep8_c64: frontend: type: vllm-router args: @@ -237,56 +203,21 @@ override_dep4_c48: setup_script: vllm-router-0.1.14.sh roles: agg: - gpus: 4 - args: - tensor-parallel-size: 1 - data-parallel-size: 4 - enable-expert-parallel: true - enable-ep-weight-filter: true - moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 - long-prefill-token-threshold: 512 - max-num-batched-tokens: 8192 - gpu-memory-utilization: 0.95 - max-num-seqs: 24 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356000000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' - env: - PYTORCH_ALLOC_CONF: 'expandable_segments:True' - PYTHONHASHSEED: '42' - benchmark: - env: - CONC: '48' - KV_OFFLOADING: dram - TOTAL_CPU_DRAM_GB: '1424' - AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' - -override_dep4_c64: - frontend: - type: vllm-router - args: - policy: consistent_hash - prometheus-host: 127.0.0.1 - prometheus-port: 18000 - request-timeout-secs: 14400 - disable-retries: true - setup_script: vllm-router-0.1.14.sh - roles: - agg: - gpus: 4 + gpus: 8 args: tensor-parallel-size: 1 - data-parallel-size: 4 + data-parallel-size: 8 enable-expert-parallel: true enable-ep-weight-filter: true moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 + prefill-schedule-interval: 16 long-prefill-token-threshold: 512 - max-num-batched-tokens: 8192 - gpu-memory-utilization: 0.95 - max-num-seqs: 32 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356000000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' + kv-cache-memory-bytes: 107000000000 + max-num-batched-tokens: 8304 + gpu-memory-utilization: 0.92 + max-num-seqs: 16 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,105,112]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' env: PYTORCH_ALLOC_CONF: 'expandable_segments:True' PYTHONHASHSEED: '42' @@ -294,10 +225,10 @@ override_dep4_c64: env: CONC: '64' KV_OFFLOADING: dram - TOTAL_CPU_DRAM_GB: '1424' + TOTAL_CPU_DRAM_GB: '2849' AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' -override_dep8_c128: +override_dep8_c96: frontend: type: vllm-router args: @@ -316,24 +247,25 @@ override_dep8_c128: enable-expert-parallel: true enable-ep-weight-filter: true moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 + prefill-schedule-interval: 16 long-prefill-token-threshold: 512 - max-num-batched-tokens: 16384 + kv-cache-memory-bytes: 107000000000 + max-num-batched-tokens: 8360 gpu-memory-utilization: 0.92 - max-num-seqs: 32 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' + max-num-seqs: 24 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,102,105,108,112,114,119,120,126,132,133,138,140,144,147,154,161,168]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' env: PYTORCH_ALLOC_CONF: 'expandable_segments:True' PYTHONHASHSEED: '42' benchmark: env: - CONC: '128' + CONC: '96' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2849' AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' -override_dep8_c256: +override_dep8_c128: frontend: type: vllm-router args: @@ -352,24 +284,25 @@ override_dep8_c256: enable-expert-parallel: true enable-ep-weight-filter: true moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 + prefill-schedule-interval: 16 long-prefill-token-threshold: 512 - max-num-batched-tokens: 16384 + kv-cache-memory-bytes: 107000000000 + max-num-batched-tokens: 8416 gpu-memory-utilization: 0.92 - max-num-seqs: 64 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' + max-num-seqs: 32 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,102,105,108,112,114,119,120,126,132,133,138,140,144,147,150,154,156,161,162,168,174,175,180,182,186,189,192,196,203,210,217,224]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' env: PYTORCH_ALLOC_CONF: 'expandable_segments:True' PYTHONHASHSEED: '42' benchmark: env: - CONC: '256' + CONC: '128' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2849' AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' -override_dep8_c384: +override_dep8_c192: frontend: type: vllm-router args: @@ -388,24 +321,25 @@ override_dep8_c384: enable-expert-parallel: true enable-ep-weight-filter: true moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 + prefill-schedule-interval: 16 long-prefill-token-threshold: 512 - max-num-batched-tokens: 16384 + kv-cache-memory-bytes: 105000000000 + max-num-batched-tokens: 8528 gpu-memory-utilization: 0.92 - max-num-seqs: 96 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,292,296,300,304,308,312,316,320,324,328,332,336,340,344,348,352,356,360,364,368,372,376,380,384],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' + max-num-seqs: 48 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,102,105,108,112,114,119,120,126,132,133,138,140,144,147,150,154,156,161,162,168,174,175,180,182,186,189,192,196,198,203,204,210,216,217,222,224,228,231,234,238,240,245,246,252,258,259,264,266,270,273,276,280,282,287,288,294,301,308,315,322,329,336]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' env: PYTORCH_ALLOC_CONF: 'expandable_segments:True' PYTHONHASHSEED: '42' benchmark: env: - CONC: '384' + CONC: '192' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2849' AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' -override_dep8_c512: +override_dep8_c256: frontend: type: vllm-router args: @@ -424,24 +358,25 @@ override_dep8_c512: enable-expert-parallel: true enable-ep-weight-filter: true moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 + prefill-schedule-interval: 16 long-prefill-token-threshold: 512 - max-num-batched-tokens: 16384 + kv-cache-memory-bytes: 105000000000 + max-num-batched-tokens: 8640 gpu-memory-utilization: 0.92 - max-num-seqs: 128 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,292,296,300,304,308,312,316,320,324,328,332,336,340,344,348,352,356,360,364,368,372,376,380,384,388,392,396,400,404,408,412,416,420,424,428,432,436,440,444,448,452,456,460,464,468,472,476,480,484,488,492,496,500,504,508,512],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' + max-num-seqs: 64 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,102,105,108,112,114,119,120,126,132,133,138,140,144,147,150,154,156,161,162,168,174,175,180,182,186,189,192,196,198,203,204,210,216,217,222,224,228,231,234,238,240,245,246,252,258,259,264,266,270,273,276,280,282,287,288,294,300,301,306,308,312,315,318,322,324,329,330,336,342,343,348,350,354,357,360,364,366,371,372,378,384,385,392,399,406,413,420,427,434,441,448]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' env: PYTORCH_ALLOC_CONF: 'expandable_segments:True' PYTHONHASHSEED: '42' benchmark: env: - CONC: '512' + CONC: '256' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2849' AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' -override_dep8_c576: +override_dep8_c288: frontend: type: vllm-router args: @@ -460,19 +395,20 @@ override_dep8_c576: enable-expert-parallel: true enable-ep-weight-filter: true moe-backend: deep_gemm_amxf4_mega_moe - prefill-schedule-interval: 8 + prefill-schedule-interval: 16 long-prefill-token-threshold: 512 - max-num-batched-tokens: 16384 + kv-cache-memory-bytes: 105000000000 + max-num-batched-tokens: 8696 gpu-memory-utilization: 0.92 - max-num-seqs: 144 - compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,288,292,296,300,304,308,312,316,320,324,328,332,336,340,344,348,352,356,360,364,368,372,376,380,384,388,392,396,400,404,408,412,416,420,424,428,432,436,440,444,448,452,456,460,464,468,472,476,480,484,488,492,496,500,504,508,512,516,520,524,528,532,536,540,544,548,552,556,560,564,568,572,576],"mode":0}' - kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' + max-num-seqs: 72 + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[6,7,12,14,18,21,24,28,30,35,36,42,48,49,54,56,60,63,66,70,72,77,78,84,90,91,96,98,102,105,108,112,114,119,120,126,132,133,138,140,144,147,150,154,156,161,162,168,174,175,180,182,186,189,192,196,198,203,204,210,216,217,222,224,228,231,234,238,240,245,246,252,258,259,264,266,270,273,276,280,282,287,288,294,300,301,306,308,312,315,318,322,324,329,330,336,342,343,348,350,354,357,360,364,366,371,372,378,384,385,390,392,396,399,402,406,408,413,414,420,426,427,432,434,441,448,455,462,469,476,483,490,497,504]}' + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":356125000000,"enable_cross_layers_blocks":"true","lazy_offload":false}}' env: PYTORCH_ALLOC_CONF: 'expandable_segments:True' PYTHONHASHSEED: '42' benchmark: env: - CONC: '576' + CONC: '288' KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2849' AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID: '1' diff --git a/inferencex-e2e/configs/nvidia-master.yaml b/inferencex-e2e/configs/nvidia-master.yaml index b0fc9ef451..46976f0f0f 100644 --- a/inferencex-e2e/configs/nvidia-master.yaml +++ b/inferencex-e2e/configs/nvidia-master.yaml @@ -1645,8 +1645,8 @@ dsv4-fp8-h200-dynamo-sglang-agentic-agg: additional-settings: - "CONFIG_FILE=recipes/dsv4/sglang/h200-fp8/agentx/agg-tp8-mtp-kvoffload.yaml" dsv4-fp4-b300-vllm-agentic-mtp: - image: vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-426e59f - model: deepseek-ai/DeepSeek-V4-Pro + image: vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-20ebb28 + model: deepseek-ai/DeepSeek-V4-Pro-0813 model-prefix: dsv4 runner: cluster:b300-dsxe precision: fp4 @@ -1656,16 +1656,12 @@ dsv4-fp4-b300-vllm-agentic-mtp: agentic-coding: - dram-utilization: 0.95 search-space: - # TP8 GPU-resident + MTP (num_speculative_tokens=3) - - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } - # TP4 GPU-resident + MTP (num_speculative_tokens=3) - - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 4, 6, 8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } - # TP4 SimpleCPU + MTP (num_speculative_tokens=3) - - { tp: 4, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [16], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } - # DEP4 SimpleCPU + MTP (num_speculative_tokens=3) - - { tp: 4, ep: 4, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [48, 64], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } - # DEP8 SimpleCPU + MTP (num_speculative_tokens=3) - - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, spec-decoding: mtp, conc-list: [128, 256, 384, 512, 576], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } + # TP8 GPU-resident + DSpark6 + - { tp: 8, kv-offloading: none, spec-decoding: draft_model, conc-list: [1, 2, 4, 8, 16], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } + # TP4 GPU-resident + DSpark6 + - { tp: 4, kv-offloading: none, spec-decoding: draft_model, conc-list: [8], srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } + # DEP8 SimpleCPU + DSpark6 + - { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, spec-decoding: draft_model, conc-list: [64, 96, 128, 192, 256, 288], router: { name: vllm-router, version: "0.1.14" }, srt-recipe: benchmarks/single_node/srt-slurm-recipes/dsv4/vllm/b300-fp4-mtp/agentic.yaml } qwen3.5-fp8-h200-sglang: image: lmsysorg/sglang:v0.5.14-cu130 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e280ad359f..8c59647537 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -8995,3 +8995,11 @@ - "Validate MiniMax-M3 AgentX on H200 with vLLM FP8 after moving the end-to-end project into inferencex-e2e/. Preserve the existing image, recipe, and benchmark settings." - "将端到端项目迁入 inferencex-e2e/ 后,验证 H200 上的 vLLM FP8 MiniMax-M3 AgentX 运行路径,沿用现有镜像、配方和基准测试设置。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3525 + +- config-keys: + - dsv4-fp4-b300-vllm-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Update B300 vLLM AgentX to DSpark6 on a new image with a sampled concurrency grid and per-mode --kv-cache-memory-bytes pins." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3297 diff --git a/inferencex-e2e/runners/launch_b300-dsxe.sh b/inferencex-e2e/runners/launch_b300-dsxe.sh index efd0506131..f1f4764373 100755 --- a/inferencex-e2e/runners/launch_b300-dsxe.sh +++ b/inferencex-e2e/runners/launch_b300-dsxe.sh @@ -27,7 +27,7 @@ if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash && printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh" } > "$BATCH_SCRIPT" BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" - --nodes=1 --ntasks=1 --gres="gpu:$GPU_COUNT" --exclusive --mem=0 + --nodes=1 --ntasks=1 --cpus-per-task=192 --gres="gpu:$GPU_COUNT" --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --job-name="$RUNNER_NAME" --export=ALL --chdir="$GITHUB_WORKSPACE" --output="$BATCH_LOG") if [[ -n "${SALLOC_EXCLUDE:-}" ]]; then @@ -137,7 +137,8 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then SRT_MODEL_PATH="$MODEL_ROOT/${MODEL##*/}" if [[ "$MODEL" == nvidia/DeepSeek-R1-0528-FP4-V2 ]]; then SRT_MODEL_PATH="$MODEL_ROOT/DeepSeek-R1-0528-NVFP4-v2" - elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || "${MODEL##*/}" == DeepSeek-V4-Pro-0813 ]]; then + elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || + ( "${MODEL##*/}" == DeepSeek-V4-Pro-0813 && "$FRAMEWORK" != vllm ) ]]; then # Not staged on every node's NVMe; read the shared copy. SRT_MODEL_PATH="$SHARED_MODEL_ROOT/${MODEL##*/}" fi From 02f4b1b022f5996482510df840ce64e8947d988d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Mon, 28 Sep 2026 10:23:46 -0500 Subject: [PATCH 2/2] revert(b300): keep the launcher unchanged for DSpark6 The sbatch wrapper only serves DSV4.1 Flash SGLang, and the recipe already requests 192 CPUs through sbatch_directives. The 0813 checkpoint loads from the shared model root as before. --- inferencex-e2e/runners/launch_b300-dsxe.sh | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/runners/launch_b300-dsxe.sh b/inferencex-e2e/runners/launch_b300-dsxe.sh index f1f4764373..efd0506131 100755 --- a/inferencex-e2e/runners/launch_b300-dsxe.sh +++ b/inferencex-e2e/runners/launch_b300-dsxe.sh @@ -27,7 +27,7 @@ if [[ "$IS_MULTINODE" != true && "${MODEL_PREFIX:-}" == dsv41flash && printf '%q\n' "$GITHUB_WORKSPACE/runners/launch_b300-dsxe.sh" } > "$BATCH_SCRIPT" BATCH_ARGS=(--parsable --partition="$SLURM_PARTITION" --account="$SLURM_ACCOUNT" - --nodes=1 --ntasks=1 --cpus-per-task=192 --gres="gpu:$GPU_COUNT" --exclusive --mem=0 + --nodes=1 --ntasks=1 --gres="gpu:$GPU_COUNT" --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --job-name="$RUNNER_NAME" --export=ALL --chdir="$GITHUB_WORKSPACE" --output="$BATCH_LOG") if [[ -n "${SALLOC_EXCLUDE:-}" ]]; then @@ -137,8 +137,7 @@ if [[ "$EXECUTION_PATH" == native-single-node ]]; then SRT_MODEL_PATH="$MODEL_ROOT/${MODEL##*/}" if [[ "$MODEL" == nvidia/DeepSeek-R1-0528-FP4-V2 ]]; then SRT_MODEL_PATH="$MODEL_ROOT/DeepSeek-R1-0528-NVFP4-v2" - elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || - ( "${MODEL##*/}" == DeepSeek-V4-Pro-0813 && "$FRAMEWORK" != vllm ) ]]; then + elif [[ " ${STAGED_MODELS[*]} " != *" ${MODEL##*/} "* || "${MODEL##*/}" == DeepSeek-V4-Pro-0813 ]]; then # Not staged on every node's NVMe; read the shared copy. SRT_MODEL_PATH="$SHARED_MODEL_ROOT/${MODEL##*/}" fi