diff --git a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh index 0f6a801136..78a60cec69 100644 --- a/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh +++ b/benchmarks/single_node/fixed_seq_len/dsr1_fp8_mi325x.sh @@ -37,7 +37,7 @@ python3 -m sglang.launch_server \ --model-path=$MODEL --host=0.0.0.0 --port=$PORT --trust-remote-code \ --tensor-parallel-size=$TP \ --mem-fraction-static=0.8 \ ---cuda-graph-max-bs=128 \ +--cuda-graph-max-bs-decode=128 \ --chunked-prefill-size=131072 \ --num-continuous-decode-steps=4 \ --max-prefill-tokens=131072 \ diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 9b4fcadeb7..11a38ee897 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -78,7 +78,7 @@ dsr1-fp8-mi300x-sglang: - { tp: 8, conc-start: 4, conc-end: 64 } dsr1-fp8-mi325x-sglang: - image: lmsysorg/sglang:v0.5.19-rocm700-mi30x + image: lmsysorg/sglang:v0.5.20-rocm720-mi30x model: deepseek-ai/DeepSeek-R1-0528 model-prefix: dsr1 runner: mi325x diff --git a/perf-changelog.yaml b/perf-changelog.yaml index c5b4309f5f..35e5afc5c8 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8455,3 +8455,9 @@ description: - "Update B200 vLLM AgentX to DSpark6 and a new image with TP8 and DEP8 configurations." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3274 + +- config-keys: + - dsr1-fp8-mi325x-sglang + description: + - "Update SGLang image from v0.5.19-rocm700-mi30x to v0.5.20-rocm720-mi30x and rename the CUDA-graph max-bs flag." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3318