From 7e90a18b3b285b24d112113ebc52149f9c5a02a3 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:53:41 -0400 Subject: [PATCH 1/6] feat(h100): add TensorRT-LLM ModelScope coverage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 H100 添加 TensorRT-LLM ModelScope 覆盖,使用 Qwen3-0.6B BF16 和与 1.3.0rc27 镜像源码匹配的回移补丁。 --- .../fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh | 126 +++++++++ configs/nvidia-master.yaml | 17 ++ perf-changelog.yaml | 7 + runners/patch_trtllm_modelscope.py | 241 ++++++++++++++++++ 4 files changed, 391 insertions(+) create mode 100755 benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh create mode 100755 runners/patch_trtllm_modelscope.py diff --git a/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh b/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh new file mode 100755 index 0000000000..611df2f3cf --- /dev/null +++ b/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh @@ -0,0 +1,126 @@ +#!/usr/bin/env bash + +source "$(dirname "$0")/../../benchmark_lib.sh" + +check_env_vars \ + MODEL \ + TP \ + CONC \ + ISL \ + OSL \ + RANDOM_RANGE_RATIO \ + RESULT_FILENAME \ + EVAL_ONLY \ + RUN_EVAL \ + PORT \ + HF_HUB_CACHE + +if [[ -n "$SLURM_JOB_ID" ]]; then + echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" +fi + +python3 - <<'PY' +from importlib.metadata import version + +expected = "1.3.0rc27" +actual = version("tensorrt_llm") +if actual != expected: + raise SystemExit( + f"Expected TensorRT-LLM {expected} for the pinned NGC image, got {actual}" + ) +PY + +python3 -m pip install --quiet --disable-pip-version-check \ + "modelscope==1.40.1" "modelscope-hub==0.4.3" +python3 "$(dirname "$0")/../../../runners/patch_trtllm_modelscope.py" + +export TRTLLM_USE_MODELSCOPE=true +export MODELSCOPE_CACHE="$HF_HUB_CACHE/modelscope" + +# Resolve through TensorRT-LLM's patched hub boundary on the H100 node. Keep +# serving the remote model ID below so model loading, config, and tokenizer +# paths all exercise the ModelScope integration. +MODEL_PATH_FILE=$(mktemp) +python3 - "$MODEL" "$MODEL_PATH_FILE" <<'PY' +import sys +from pathlib import Path + +from tensorrt_llm.llmapi.utils import download_hf_model + +model_path = download_hf_model(sys.argv[1]) +Path(sys.argv[2]).write_text(str(model_path), encoding="utf-8") +PY +MODEL_PATH=$(<"$MODEL_PATH_FILE") +rm -f "$MODEL_PATH_FILE" +export MODEL_PATH + +if [[ ! -f "$MODEL_PATH/config.json" ]]; then + echo "ModelScope snapshot is missing config.json: $MODEL_PATH" >&2 + exit 1 +fi + +echo "ModelScope snapshot: $MODEL_PATH" +echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL" +nvidia-smi + +SERVER_LOG=/workspace/server.log +EXTRA_CONFIG_FILE=$(mktemp --suffix=.yaml) +MAX_BATCH_SIZE=$((CONC > 16 ? CONC : 16)) +MAX_MODEL_LEN=$((ISL + OSL + 256)) +MAX_NUM_TOKENS=$((((ISL + CONC + 127) / 128) * 128)) +MAX_NUM_TOKENS=$((MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192)) + +cat > "$EXTRA_CONFIG_FILE" < "$SERVER_LOG" 2>&1 & + +SERVER_PID=$! + +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +run_benchmark_serving \ + --model "$MODEL" \ + --port "$PORT" \ + --backend openai \ + --input-len "$ISL" \ + --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$((CONC * 10))" \ + --max-concurrency "$CONC" \ + --result-filename "$RESULT_FILENAME" \ + --result-dir /workspace/ + +if [[ "$RUN_EVAL" == "true" ]]; then + run_eval --framework lm-eval --port "$PORT" + append_lm_eval_summary +fi + +stop_gpu_monitor +rm -f "$EXTRA_CONFIG_FILE" +set +x diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index c5d5edac68..fdc754285d 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -3953,6 +3953,23 @@ qwen3.5-fp8-h100-sglang: - { tp: 8, ep: 1, conc-start: 1, conc-end: 8 } - { tp: 8, ep: 8, conc-start: 16, conc-end: 256 } +# ModelScope integration coverage for TensorRT-LLM. The image version maps to +# NVIDIA/TensorRT-LLM tag v1.3.0rc27 at commit 6e1cc953c071b8a9055b03ef2ae4ee0bc4c645c4. +qwen3-0.6b-bf16-h100-trt-modelscope: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc27 + model: Qwen/Qwen3-0.6B + model-prefix: qwen3-0.6b + runner: cluster:h100-dgxc + precision: bf16 + framework: trt + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 1, conc-list: [1, 4, 16, 32, 64] } + qwen3.5-fp8-h100-sglang-mtp: image: lmsysorg/sglang:v0.5.19-cu130 model: Qwen/Qwen3.5-397B-A17B-FP8 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index c5b4309f5f..e87d36a562 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8455,3 +8455,10 @@ description: - "Update B200 vLLM AgentX to DSpark6 and a new image with TP8 and DEP8 configurations." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3274 + +- config-keys: + - qwen3-0.6b-bf16-h100-trt-modelscope + description: + - "Add H100 TensorRT-LLM 1.3.0rc27 coverage for Qwen3-0.6B in BF16, resolving the model through ModelScope with a source-matched runtime backport of SemiAnalysisAI/TensorRT-LLM#2." + - "为 Qwen3-0.6B BF16 添加 H100 TensorRT-LLM 1.3.0rc27 覆盖,通过 ModelScope 解析模型,并使用与镜像源码匹配的 SemiAnalysisAI/TensorRT-LLM#2 运行时回移补丁。" + pr-link: TBD diff --git a/runners/patch_trtllm_modelscope.py b/runners/patch_trtllm_modelscope.py new file mode 100755 index 0000000000..64bcc92c6d --- /dev/null +++ b/runners/patch_trtllm_modelscope.py @@ -0,0 +1,241 @@ +#!/usr/bin/env python3 +"""Backport ModelScope loading to the pinned TensorRT-LLM runtime.""" + +from __future__ import annotations + +import importlib.util +import sys +from pathlib import Path + +HF_IMPORT = "from huggingface_hub import snapshot_download\n" +ALIASED_HF_IMPORT = ( + "from huggingface_hub import snapshot_download as hf_snapshot_download\n" +) + +OLD_DOWNLOAD_BLOCK = '''def download_hf_model(model: str, revision: Optional[str] = None) -> Path: + ignore_patterns = ["original/**/*"] + logger.info(f"Downloading model {model} from HuggingFace") + with get_file_lock(model): + hf_folder = snapshot_download( + model, + local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE, + ignore_patterns=ignore_patterns, + revision=revision, + tqdm_class=DisabledTqdm) + logger.info(f"Finished downloading model {model} from HuggingFace") + return Path(hf_folder) + + +def download_hf_partial(model: str, + allow_patterns: List[str], + revision: Optional[str] = None) -> Path: + """Download a partial model from HuggingFace. + + Args: + model: The model name or path. + revision: The revision to use for the model. + allow_patterns: The patterns to allow for the model. + + Returns: + The path to the downloaded model. + """ + with get_file_lock(model): + hf_folder = snapshot_download( + model, + local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE, + revision=revision, + allow_patterns=allow_patterns, + tqdm_class=DisabledTqdm) + return Path(hf_folder) + + +''' + +NEW_DOWNLOAD_BLOCK = '''def download_hf_model(model: str, revision: Optional[str] = None) -> Path: + ignore_patterns = ["original/**/*"] + hub_name = "ModelScope" if use_modelscope() else "Hugging Face" + logger.info(f"Downloading model {model} from {hub_name}") + with get_file_lock(model): + model_folder = _snapshot_download(model, + ignore_patterns=ignore_patterns, + revision=revision) + logger.info(f"Finished downloading model {model} from {hub_name}") + return Path(model_folder) + + +def download_hf_partial(model: str, + allow_patterns: List[str], + revision: Optional[str] = None) -> Path: + """Download selected model files from the configured model hub. + + Args: + model: The model name or path. + revision: The revision to use for the model. + allow_patterns: The patterns to allow for the model. + + Returns: + The path to the downloaded model. + """ + with get_file_lock(model): + model_folder = _snapshot_download(model, + revision=revision, + allow_patterns=allow_patterns) + return Path(model_folder) + + +def use_modelscope() -> bool: + """Return whether remote model IDs should resolve through ModelScope.""" + return os.environ.get("TRTLLM_USE_MODELSCOPE", "false").strip().lower( + ) in ("1", "true") + + +def _snapshot_download(model: str, + revision: Optional[str] = None, + ignore_patterns: Optional[List[str]] = None, + allow_patterns: Optional[List[str]] = None) -> str: + """Download a snapshot from ModelScope or Hugging Face. + + ModelScope uses different names for its file filters. Keep the optional + import in this boundary so standard TensorRT-LLM installations do not need + the ``modelscope`` package. + """ + local_files_only = huggingface_hub.constants.HF_HUB_OFFLINE + if use_modelscope(): + try: + from modelscope.hub.snapshot_download import snapshot_download + except ImportError as error: + raise ImportError( + "TRTLLM_USE_MODELSCOPE is enabled, but ModelScope is not " + "installed. Install it with `pip install modelscope`.") from error + + kwargs = { + "model_id": model, + "local_files_only": local_files_only, + "revision": revision, + } + if ignore_patterns: + kwargs["ignore_file_pattern"] = ignore_patterns + if allow_patterns: + kwargs["allow_file_pattern"] = allow_patterns + return snapshot_download(**kwargs) + + return hf_snapshot_download( + model, + local_files_only=local_files_only, + ignore_patterns=ignore_patterns, + allow_patterns=allow_patterns, + revision=revision, + tqdm_class=DisabledTqdm) + + +''' + + +def installed_package_root() -> Path: + """Return the installed TensorRT-LLM package root.""" + spec = importlib.util.find_spec("tensorrt_llm") + if spec is None or not spec.submodule_search_locations: + raise RuntimeError("tensorrt_llm package is not installed") + return Path(next(iter(spec.submodule_search_locations))) + + +def replace_once(source: str, old: str, new: str, path: Path) -> str: + """Replace one pinned source fragment or fail on an unknown runtime.""" + count = source.count(old) + if count != 1: + raise RuntimeError( + f"unsupported TensorRT-LLM source at {path}: " + f"expected one pinned fragment, found {count}" + ) + return source.replace(old, new, 1) + + +def patch_utils(path: Path) -> bool: + """Patch the hub download boundary.""" + source = path.read_text(encoding="utf-8") + if "def use_modelscope() -> bool:" in source: + if ALIASED_HF_IMPORT not in source: + raise RuntimeError(f"partial ModelScope patch found at {path}") + return False + + source = replace_once(source, HF_IMPORT, ALIASED_HF_IMPORT, path) + source = replace_once(source, OLD_DOWNLOAD_BLOCK, NEW_DOWNLOAD_BLOCK, path) + path.write_text(source, encoding="utf-8") + return True + + +def patch_llm(path: Path) -> bool: + """Load tokenizer and configuration from the downloaded snapshot.""" + source = path.read_text(encoding="utf-8") + tokenizer_marker = " model_path = self._hf_model_dir or self.args.model\n" + generation_marker = " model_dir = self._hf_model_dir or self.args.model\n" + if tokenizer_marker in source: + if source.count(generation_marker) < 3: + raise RuntimeError(f"partial ModelScope path patch found at {path}") + return False + + tokenizer_start = source.index(" def _try_load_tokenizer(") + tokenizer_end = source.index("\n @property\n def tokenizer", tokenizer_start) + tokenizer_source = source[tokenizer_start:tokenizer_end] + anchor = ( + " if self.args.tokenizer is not None:\n" + " assert isinstance(self.args.tokenizer, TokenizerBase)\n" + " return self.args.tokenizer\n\n" + ) + if tokenizer_source.count("self.args.model") != 4: + raise RuntimeError( + f"unsupported TensorRT-LLM tokenizer loader at {path}: " + "unexpected model-path reference count" + ) + tokenizer_source = tokenizer_source.replace("self.args.model", "model_path") + tokenizer_source = replace_once( + tokenizer_source, anchor, anchor + tokenizer_marker + "\n", path + ) + source = source[:tokenizer_start] + tokenizer_source + source[tokenizer_end:] + + source = replace_once( + source, + " return ModelLoader.load_hf_generation_config(self.args.model)\n", + generation_marker + + " return ModelLoader.load_hf_generation_config(model_dir)\n", + path, + ) + source = replace_once( + source, + " return ModelLoader.load_hf_model_config(\n" + " self.args.model, trust_remote_code=self.args.trust_remote_code)\n", + generation_marker + " return ModelLoader.load_hf_model_config(\n" + " model_dir, trust_remote_code=self.args.trust_remote_code)\n", + path, + ) + path.write_text(source, encoding="utf-8") + return True + + +def main(argv: list[str]) -> int: + if len(argv) > 2: + print(f"Usage: {argv[0]} [TENSORRT_LLM_PACKAGE_ROOT]", file=sys.stderr) + return 2 + + try: + package_root = ( + Path(argv[1]).resolve() if len(argv) == 2 else installed_package_root() + ) + changed = [ + patch_utils(package_root / "llmapi/utils.py"), + patch_llm(package_root / "llmapi/llm.py"), + ] + except (OSError, RuntimeError, ValueError) as error: + print( + f"ERROR: failed to patch TensorRT-LLM ModelScope support: {error}", + file=sys.stderr, + ) + return 1 + + state = "Patched" if any(changed) else "Already patched" + print(f"{state} TensorRT-LLM ModelScope support") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) From 3749056cb6de69f22e4102bf27db1b35303ad1fc Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:53:58 -0400 Subject: [PATCH 2/6] docs: record ModelScope patch waiver MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 记录 PR #3323 的 TensorRT-LLM ModelScope 运行时补丁豁免和移除计划。 --- docs/waiver/3323.md | 49 +++++++++++++++++++++++++++++++++++++++++++++ perf-changelog.yaml | 2 +- 2 files changed, 50 insertions(+), 1 deletion(-) create mode 100644 docs/waiver/3323.md diff --git a/docs/waiver/3323.md b/docs/waiver/3323.md new file mode 100644 index 0000000000..e65e1f9829 --- /dev/null +++ b/docs/waiver/3323.md @@ -0,0 +1,49 @@ +# Inference-engine patch waiver — PR #3323 + +Filed per [`docs/PR_REVIEW_CHECKLIST.md`](../PR_REVIEW_CHECKLIST.md): this PR patches the pinned +TensorRT-LLM image before serving because the released image predates ModelScope model loading. + +## Config covered + +- **Master config entry:** `qwen3-0.6b-bf16-h100-trt-modelscope` in + [`configs/nvidia-master.yaml`](../../configs/nvidia-master.yaml) +- **Pinned image:** `nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc27` +- **Image source:** NVIDIA/TensorRT-LLM tag `v1.3.0rc27`, commit + `6e1cc953c071b8a9055b03ef2ae4ee0bc4c645c4` +- **Patch entrypoint:** + [`runners/patch_trtllm_modelscope.py`](../../runners/patch_trtllm_modelscope.py), invoked by + [`benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh`](../../benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh) + +## What is patched + +The patcher backports the ModelScope integration from +[SemiAnalysisAI/TensorRT-LLM#2](https://github.com/SemiAnalysisAI/TensorRT-LLM/pull/2) to the two +installed Python modules that participate in this benchmark: + +- `tensorrt_llm/llmapi/utils.py` routes full and partial snapshot downloads through ModelScope when + `TRTLLM_USE_MODELSCOPE=true`, while preserving Hugging Face as the default. +- `tensorrt_llm/llmapi/llm.py` loads the tokenizer, generation config, and model config from the + resolved local snapshot rather than retrying the remote Hugging Face model ID. + +The backport is source-matched to `v1.3.0rc27`, exact-anchor gated, and idempotent. It refuses an +unknown or partially patched installed source tree. `modelscope==1.40.1` and +`modelscope-hub==0.4.3` are installed in the H100 container before the patch is applied. + +## Why the unmodified upstream image cannot run this benchmark + +TensorRT-LLM `1.3.0rc27` resolves remote model IDs exclusively with `huggingface_hub`. It has no +ModelScope switch or downloader and subsequently loads tokenizer and configuration files from the +original remote ID. Therefore the stock image cannot validate TensorRT-LLM model loading from +ModelScope for `Qwen/Qwen3-0.6B`; installing the optional ModelScope dependency alone does not change +that behavior. + +## Upstream PR + +- https://github.com/SemiAnalysisAI/TensorRT-LLM/pull/2 + +## Removal plan + +Once an NGC TensorRT-LLM release includes the ModelScope integration, update +`qwen3-0.6b-bf16-h100-trt-modelscope` to the first matching release image and verify its source tag. +In the same PR, remove `runners/patch_trtllm_modelscope.py`, remove its invocation and runtime package +installation from the benchmark script, and delete this waiver. diff --git a/perf-changelog.yaml b/perf-changelog.yaml index e87d36a562..72a1ec561c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8461,4 +8461,4 @@ description: - "Add H100 TensorRT-LLM 1.3.0rc27 coverage for Qwen3-0.6B in BF16, resolving the model through ModelScope with a source-matched runtime backport of SemiAnalysisAI/TensorRT-LLM#2." - "为 Qwen3-0.6B BF16 添加 H100 TensorRT-LLM 1.3.0rc27 覆盖,通过 ModelScope 解析模型,并使用与镜像源码匹配的 SemiAnalysisAI/TensorRT-LLM#2 运行时回移补丁。" - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3323 From 8fb22ef5122784357257d2fb2e3752b8ff9ba26d Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 20 Sep 2026 16:03:51 -0400 Subject: [PATCH 3/6] fix(modelscope): preserve snapshot glob filters MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 改用 ModelScope 原生 glob 过滤参数,并让基准客户端复用 ModelScope 本地 tokenizer。 --- .../fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh | 3 ++- runners/patch_trtllm_modelscope.py | 12 ++++++------ 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh b/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh index 611df2f3cf..f764bf5900 100755 --- a/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh +++ b/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh @@ -8,6 +8,7 @@ check_env_vars \ CONC \ ISL \ OSL \ + MAX_MODEL_LEN \ RANDOM_RANGE_RATIO \ RESULT_FILENAME \ EVAL_ONLY \ @@ -66,7 +67,6 @@ nvidia-smi SERVER_LOG=/workspace/server.log EXTRA_CONFIG_FILE=$(mktemp --suffix=.yaml) MAX_BATCH_SIZE=$((CONC > 16 ? CONC : 16)) -MAX_MODEL_LEN=$((ISL + OSL + 256)) MAX_NUM_TOKENS=$((((ISL + CONC + 127) / 128) * 128)) MAX_NUM_TOKENS=$((MAX_NUM_TOKENS > 8192 ? MAX_NUM_TOKENS : 8192)) @@ -106,6 +106,7 @@ wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$S run_benchmark_serving \ --model "$MODEL" \ + --tokenizer "$MODEL_PATH" \ --port "$PORT" \ --backend openai \ --input-len "$ISL" \ diff --git a/runners/patch_trtllm_modelscope.py b/runners/patch_trtllm_modelscope.py index 64bcc92c6d..f979c840ca 100755 --- a/runners/patch_trtllm_modelscope.py +++ b/runners/patch_trtllm_modelscope.py @@ -95,9 +95,8 @@ def _snapshot_download(model: str, allow_patterns: Optional[List[str]] = None) -> str: """Download a snapshot from ModelScope or Hugging Face. - ModelScope uses different names for its file filters. Keep the optional - import in this boundary so standard TensorRT-LLM installations do not need - the ``modelscope`` package. + Keep the optional import in this boundary so standard TensorRT-LLM + installations do not need the ``modelscope`` package. """ local_files_only = huggingface_hub.constants.HF_HUB_OFFLINE if use_modelscope(): @@ -106,7 +105,8 @@ def _snapshot_download(model: str, except ImportError as error: raise ImportError( "TRTLLM_USE_MODELSCOPE is enabled, but ModelScope is not " - "installed. Install it with `pip install modelscope`.") from error + "installed. Install it with `pip install 'modelscope>=1.20'`." + ) from error kwargs = { "model_id": model, @@ -114,9 +114,9 @@ def _snapshot_download(model: str, "revision": revision, } if ignore_patterns: - kwargs["ignore_file_pattern"] = ignore_patterns + kwargs["ignore_patterns"] = ignore_patterns if allow_patterns: - kwargs["allow_file_pattern"] = allow_patterns + kwargs["allow_patterns"] = allow_patterns return snapshot_download(**kwargs) return hf_snapshot_download( From 50383b08556f60b454badd44ade27c6ebca85350 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:22:46 -0400 Subject: [PATCH 4/6] fix(evals): define the Qwen3-0.6B GSM8K regression floor MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Document the full-split sample audit, external scale context, and conservative 0.60 model-specific floor while preserving deterministic evaluation settings. 为 Qwen3-0.6B 记录完整测试集分析、外部规模参考和保守的 0.60 GSM8K 下限,保持确定性评测设置不变。 Signed-off-by: functionstackx <47992694+functionstackx@users.noreply.github.com> --- docs/eval-agentx-procedures.md | 34 +++++++++++++++++++++++++++++++ docs/eval-agentx-procedures_zh.md | 26 +++++++++++++++++++++++ infx/evals/thresholds.yaml | 3 +++ 3 files changed, 63 insertions(+) diff --git a/docs/eval-agentx-procedures.md b/docs/eval-agentx-procedures.md index 06c9ef6cf9..4db916d0e1 100644 --- a/docs/eval-agentx-procedures.md +++ b/docs/eval-agentx-procedures.md @@ -162,6 +162,40 @@ python3 -m infx.evals.validate_scores \ Validation resolves the threshold in this order: `models..`, `default.`, then `--min-score` (default `0.85`). By default it checks numeric, non-stderr metrics beginning with `exact_match,`. It fails when a score is below threshold, no metric matches, a requested concurrency is absent, metadata has duplicates/invalid values, any point is marked failed, or result suffixes do not match the manifest. Current floors are authoritative in [`thresholds.yaml`](../infx/evals/thresholds.yaml). See [threshold resolution](../infx/evals/validate_scores.py#L61-L69) and the [validation flow](../infx/evals/validate_scores.py#L174-L302). +### Qwen3-0.6B GSM8K floor + +`models.qwen3-0.6b.gsm8k` is **0.60** for both strict-match and flexible-extract. +This is a conservative integration regression floor for the 0.6B checkpoint, not +an expected leaderboard score. The global 0.90 floor remains unchanged. + +Keep the standard five-shot chat evaluation, full 1,319-question test split, +`temperature=0`, `top_p=1`, and 5,376 generated-token limit. The +[initial H100 BF16 run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35534330261) +at concurrency 64 scored 886/1,319 (0.6717) strict and 895/1,319 (0.6785) +flexible. All responses were nonempty; 86 lacked the required numeric `####` +answer marker, 87 lacked a closing ``, and 61 repeated an identical +nonempty line of at least 30 characters five or more times. These categories +overlap. The nine-answer extraction gain does not explain most errors; inspected +wrong answers also contained arithmetic and reasoning mistakes. + +The [Qwen3 technical report, Table 8 and Section 3.3](https://arxiv.org/html/2505.09388v1) +reports 59.59% GSM8K for **Qwen3-0.6B-Base**, using four-shot chain of thought. +That is scale context only: the checkpoint and prompt differ, and it must not be +presented as a comparable five-shot chat baseline. No directly comparable +published baseline was established. The 0.60 floor is an explicit conservative +policy choice supported by that context and the inspected full-split result; +it leaves 7.17 percentage points below the observed strict score (reported +standard error 1.29 points), rather than rounding the observed score into a gate. +Validate the same fixed floor independently at concurrency 32 and 64. + +[Qwen's model guidance](https://huggingface.co/Qwen/Qwen3-0.6B#best-practices) +recommends sampling for thinking mode and warns that greedy decoding can repeat. +This integration keeps InferenceX's deterministic protocol for comparability; +the floor does not establish optimal Qwen quality or excuse request failures. +Changing thinking mode, sampling, prompts, or token budget requires a separately +documented policy and fresh full-split evidence. Preserve failed and passing +artifacts, and do not lower this floor in response to a later regression. + A manual combined throughput+eval recipe uploads eval output but the template's automatic score gate is specific to eval-only jobs. Run the validator explicitly for manual or combined runs. ## 6. Collect and inspect eval artifacts diff --git a/docs/eval-agentx-procedures_zh.md b/docs/eval-agentx-procedures_zh.md index 03c0580439..aadab4db20 100644 --- a/docs/eval-agentx-procedures_zh.md +++ b/docs/eval-agentx-procedures_zh.md @@ -162,6 +162,32 @@ python3 -m infx.evals.validate_scores \ 手动的吞吐量+eval 组合 recipe 会上传 eval 输出,但模板的自动分数 gate 专用于 eval-only 作业。对手动或组合运行必须显式执行 validator。 +### Qwen3-0.6B 的 GSM8K 下限 + +`models.qwen3-0.6b.gsm8k` 对 strict-match 和 flexible-extract 均使用 **0.60**。 +这是针对 0.6B 检查点的保守集成回归下限,不是排行榜预期分数;全局 0.90 下限保持不变。 + +保留标准五样本聊天评测、完整的 1,319 道测试题、`temperature=0`、`top_p=1` +和 5,376 个生成 token 的上限。 +[首次 H100 BF16 运行](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35534330261) +在并发 64 下取得 strict 886/1,319(0.6717)、flexible 895/1,319(0.6785)。 +所有响应均非空;86 个响应缺少所需的数字 `####` 答案标记,87 个缺少 `` +结束标记,61 个将同一条至少 30 个字符的非空行重复了五次或更多。这些类别存在重叠。 +宽松提取仅多判对九题,无法解释大多数错误;抽查的错误答案也包含算术和推理错误。 + +[Qwen3 技术报告表 8 和第 3.3 节](https://arxiv.org/html/2505.09388v1) +报告 **Qwen3-0.6B-Base** 在四样本思维链设置下的 GSM8K 分数为 59.59%。 +该结果仅用于说明模型规模背景:检查点和提示不同,不能视为可比的五样本聊天基线。 +目前未找到直接可比的已发表基线。0.60 是结合该背景和完整测试集响应检查作出的保守策略选择; +它比已观察到的 strict 分数低 7.17 个百分点(报告的标准误为 1.29 个百分点), +并非将单次分数取整后作为门槛。应在并发 32 和 64 下独立验证同一个固定下限。 + +[Qwen 模型指南](https://huggingface.co/Qwen/Qwen3-0.6B#best-practices) +建议思考模式使用采样,并警告贪心解码可能产生重复。为保持可比性,本集成沿用 +InferenceX 的确定性协议;该下限不代表 Qwen 的最佳质量,也不豁免请求失败。 +改变思考模式、采样、提示或 token 预算,需要另行记录策略并提供新的完整测试集证据。 +保留失败和成功的产物,不应因为后续回归而继续降低此下限。 + ## 6. 收集并检查 eval artifact 收集工作流会下载 `eval_*`,用 `infx/results/collect_eval_results.py` 聚合原始集合,上传 `eval_results_all/agg_eval_all.json`,并将表格写入 step summary([`collect-evals.yml`](../.github/workflows/collect-evals.yml))。 diff --git a/infx/evals/thresholds.yaml b/infx/evals/thresholds.yaml index d9b41eba40..1793887a6b 100644 --- a/infx/evals/thresholds.yaml +++ b/infx/evals/thresholds.yaml @@ -50,6 +50,9 @@ "minimaxm2.5": { "gsm8k": 0.92 }, + "qwen3-0.6b": { + "gsm8k": 0.60 + }, "qwen3.5": { "gsm8k": 0.94 } From 5b7f42559acc294af5af485d38f27700c2abf7d8 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 20 Sep 2026 19:13:01 -0400 Subject: [PATCH 5/6] test(h100): verify ModelScope downloads from empty caches MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 在每次 H100 任务中从独立空缓存启动 ModelScope 下载,并记录模型文件哈希、验证无 Hugging Face 回退。 Signed-off-by: functionstackx <47992694+functionstackx@users.noreply.github.com> --- .../fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh | 53 +++++------ docs/eval-agentx-procedures.md | 19 ++++ docs/eval-agentx-procedures_zh.md | 14 +++ perf-changelog.yaml | 7 ++ runners/modelscope_snapshot.py | 95 +++++++++++++++++++ runners/test_modelscope_snapshot.py | 73 ++++++++++++++ 6 files changed, 234 insertions(+), 27 deletions(-) create mode 100644 runners/modelscope_snapshot.py create mode 100644 runners/test_modelscope_snapshot.py diff --git a/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh b/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh index f764bf5900..721716e956 100755 --- a/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh +++ b/benchmarks/single_node/fixed_seq_len/qwen3-0.6b_bf16_h100_trt.sh @@ -1,4 +1,5 @@ #!/usr/bin/env bash +set -eo pipefail source "$(dirname "$0")/../../benchmark_lib.sh" @@ -36,31 +37,14 @@ python3 -m pip install --quiet --disable-pip-version-check \ python3 "$(dirname "$0")/../../../runners/patch_trtllm_modelscope.py" export TRTLLM_USE_MODELSCOPE=true -export MODELSCOPE_CACHE="$HF_HUB_CACHE/modelscope" +MODELSCOPE_CACHE=$(mktemp -d /tmp/modelscope-cold.XXXXXX) +COLD_HF_HOME=$(mktemp -d /tmp/modelscope-hf-empty.XXXXXX) +export MODELSCOPE_CACHE +SNAPSHOT_HELPER="$(dirname "$0")/../../../runners/modelscope_snapshot.py" +SNAPSHOT_REPORT=/workspace/modelscope_snapshot_report.json +python3 "$SNAPSHOT_HELPER" before --model "$MODEL" --cache "$MODELSCOPE_CACHE" \ + --hf-home "$COLD_HF_HOME" --report "$SNAPSHOT_REPORT" -# Resolve through TensorRT-LLM's patched hub boundary on the H100 node. Keep -# serving the remote model ID below so model loading, config, and tokenizer -# paths all exercise the ModelScope integration. -MODEL_PATH_FILE=$(mktemp) -python3 - "$MODEL" "$MODEL_PATH_FILE" <<'PY' -import sys -from pathlib import Path - -from tensorrt_llm.llmapi.utils import download_hf_model - -model_path = download_hf_model(sys.argv[1]) -Path(sys.argv[2]).write_text(str(model_path), encoding="utf-8") -PY -MODEL_PATH=$(<"$MODEL_PATH_FILE") -rm -f "$MODEL_PATH_FILE" -export MODEL_PATH - -if [[ ! -f "$MODEL_PATH/config.json" ]]; then - echo "ModelScope snapshot is missing config.json: $MODEL_PATH" >&2 - exit 1 -fi - -echo "ModelScope snapshot: $MODEL_PATH" echo "TP: $TP, CONC: $CONC, ISL: $ISL, OSL: $OSL" nvidia-smi @@ -82,15 +66,18 @@ cuda_graph_config: EOF if [[ "$EVAL_ONLY" == "true" ]]; then - setup_eval_context - MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" + # The caller supplies the model-specific context ceiling. Avoid a hub + # lookup before the server performs its cold ModelScope download. + export EVAL_MAX_MODEL_LEN="$MAX_MODEL_LEN" MAX_NUM_TOKENS="$EVAL_MAX_MODEL_LEN" fi start_gpu_monitor set -x -PYTHONNOUSERSITE=1 mpirun -n 1 --oversubscribe --allow-run-as-root \ +PYTHONNOUSERSITE=1 HF_HUB_OFFLINE=0 HF_HOME="$COLD_HF_HOME" \ + HF_HUB_CACHE="$COLD_HF_HOME/hub" HUGGINGFACE_HUB_CACHE="$COLD_HF_HOME/hub" \ + TRANSFORMERS_CACHE="$COLD_HF_HOME/hub" mpirun -n 1 --oversubscribe --allow-run-as-root \ trtllm-serve "$MODEL" --port="$PORT" \ --backend=pytorch \ --max_batch_size="$MAX_BATCH_SIZE" \ @@ -104,6 +91,18 @@ SERVER_PID=$! wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" +# Resolve only after readiness; this must reuse the fresh server download. +HF_HUB_OFFLINE=1 python3 "$SNAPSHOT_HELPER" after --model "$MODEL" --cache "$MODELSCOPE_CACHE" \ + --hf-home "$COLD_HF_HOME" --report "$SNAPSHOT_REPORT" +MODEL_PATH=$(python3 - "$SNAPSHOT_REPORT" <<'PYCODE' +import json +import sys +from pathlib import Path +print(json.loads(Path(sys.argv[1]).read_text())["snapshot"]) +PYCODE +) +export MODEL_PATH + run_benchmark_serving \ --model "$MODEL" \ --tokenizer "$MODEL_PATH" \ diff --git a/docs/eval-agentx-procedures.md b/docs/eval-agentx-procedures.md index 4db916d0e1..9b7f403b18 100644 --- a/docs/eval-agentx-procedures.md +++ b/docs/eval-agentx-procedures.md @@ -388,3 +388,22 @@ Use `scancel` or process termination only with explicit approval and a concrete - Every backend/frontend and metrics source is represented in live evidence. - Fast/smoke results are labeled diagnostic. Only the canonical candidate is used for final comparison. - Workflow and artifact collection conclude green before success is reported. + +### Qwen3-0.6B ModelScope cold-cache coverage + +The H100 ModelScope recipe starts `trtllm-serve` with the remote model ID and a +new, verified-empty ModelScope cache for every job. Its Hugging Face home/cache +is independently empty. The server performs the download; the recipe does not +predownload the weights or tokenizer. After readiness, the offline resolver must +return a snapshot within the new ModelScope cache, containing weights and the +required tokenizer/config assets. Unexpected Hugging Face cache files fail the +job; a cache version marker and the empty Transformers scaffolding files +`modules/__init__.py` and `modules/hf_remote_code.lock` are allowed. The benchmark +client uses the same resolved tokenizer path. + +`modelscope_snapshot_report.json` records the initial empty caches, resolved +snapshot path, file sizes and SHA256 hashes. Eval jobs upload this report with +their raw results; job logs also contain the report. The matrix supplies the +model-specific context ceiling before startup, avoiding a separate hub lookup. +This uses the source-matched TensorRT-LLM 1.3.0rc27 backport documented in the +engine-patch waiver. It does not build the newer TensorRT-LLM PR branch. diff --git a/docs/eval-agentx-procedures_zh.md b/docs/eval-agentx-procedures_zh.md index aadab4db20..dfa61a0ef9 100644 --- a/docs/eval-agentx-procedures_zh.md +++ b/docs/eval-agentx-procedures_zh.md @@ -378,3 +378,17 @@ gh run cancel --repo SemiAnalysisAI/InferenceX - 每个 backend/frontend 与 metrics source 都在实时证据中有所体现。 - Fast/smoke 结果明确标为诊断用途;只有 canonical candidate 用于最终比较。 - 在报告成功前,工作流与 artifact collection 均已得出 green 结论。 + +### Qwen3-0.6B ModelScope 冷缓存覆盖 + +H100 ModelScope 配方在每个任务中使用远程模型 ID 和新建、确认为空的 ModelScope +缓存启动 `trtllm-serve`,并为服务提供独立的空 Hugging Face home/cache。模型由 +服务自行下载,配方不会预下载权重或分词器。服务就绪后,离线解析器必须返回新 +ModelScope 缓存内部的快照,且其中包含权重和必需的分词器、配置文件。检测到非预期 +Hugging Face 缓存文件时任务失败;允许缓存版本标记,以及 Transformers 创建的空文件 +`modules/__init__.py` 和 `modules/hf_remote_code.lock`。基准客户端使用同一快照中的分词器。 + +`modelscope_snapshot_report.json` 记录初始空缓存、最终快照路径、文件大小和 SHA256。 +评测任务将该报告与原始结果一起上传,任务日志中也包含报告。矩阵在启动前提供模型 +上下文上限,避免另行查询模型仓库。本配方使用 engine-patch 豁免中记录的、与源码 +匹配的 TensorRT-LLM 1.3.0rc27 回移补丁,并不构建较新 TensorRT-LLM PR 分支。 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 72a1ec561c..70651cdfda 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8462,3 +8462,10 @@ - "Add H100 TensorRT-LLM 1.3.0rc27 coverage for Qwen3-0.6B in BF16, resolving the model through ModelScope with a source-matched runtime backport of SemiAnalysisAI/TensorRT-LLM#2." - "为 Qwen3-0.6B BF16 添加 H100 TensorRT-LLM 1.3.0rc27 覆盖,通过 ModelScope 解析模型,并使用与镜像源码匹配的 SemiAnalysisAI/TensorRT-LLM#2 运行时回移补丁。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3323 + +- config-keys: + - qwen3-0.6b-bf16-h100-trt-modelscope + description: + - "Exercise ModelScope cold downloads in every H100 Qwen3-0.6B job using empty, isolated hub caches and record snapshot hashes before benchmarking or evaluation." + - "每个 H100 Qwen3-0.6B 任务均使用独立空缓存测试 ModelScope 冷下载,并在基准或评测前记录模型文件哈希。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3323 diff --git a/runners/modelscope_snapshot.py b/runners/modelscope_snapshot.py new file mode 100644 index 0000000000..fcd2430cb0 --- /dev/null +++ b/runners/modelscope_snapshot.py @@ -0,0 +1,95 @@ +"""Record cold-cache and snapshot evidence for ModelScope serving.""" + +import argparse +import hashlib +import json +from pathlib import Path + + +def record_empty_caches(model: str, cache: Path, hf_home: Path, report: Path) -> None: + for directory in (cache, hf_home): + if not directory.is_dir() or any(directory.iterdir()): + raise ValueError(f"Expected an empty cache directory: {directory}") + report.write_text( + json.dumps( + { + "model": model, + "modelscope_cache": str(cache), + "hf_home": str(hf_home), + "modelscope_initial_entries": [], + "hf_initial_entries": [], + }, + indent=2, + ) + + "\n" + ) + print(f"Verified empty ModelScope cache before server startup: {cache}") + print(f"Verified empty Hugging Face home before server startup: {hf_home}") + + +def record_snapshot(model: str, cache: Path, hf_home: Path, report: Path) -> None: + from tensorrt_llm.llmapi.utils import download_hf_model + + evidence = json.loads(report.read_text()) + if (evidence["model"], evidence["modelscope_cache"], evidence["hf_home"]) != ( + model, + str(cache), + str(hf_home), + ): + raise ValueError( + "Snapshot verification does not match the cold-start configuration" + ) + snapshot = download_hf_model(model).resolve() + if not snapshot.is_relative_to(cache.resolve()): + raise ValueError(f"Snapshot is outside the fresh ModelScope cache: {snapshot}") + assets = {} + for path in sorted(snapshot.rglob("*")): + if path.is_file() and path.suffix in ( + ".json", + ".safetensors", + ".txt", + ".model", + ".jinja", + ): + with path.open("rb") as stream: + checksum = hashlib.file_digest(stream, "sha256").hexdigest() + assets[str(path.relative_to(snapshot))] = { + "bytes": path.stat().st_size, + "sha256": checksum, + } + if not {"config.json", "tokenizer_config.json", "tokenizer.json"} <= assets.keys(): + raise ValueError("The ModelScope snapshot is missing tokenizer/config files") + if not any(name.endswith(".safetensors") for name in assets): + raise ValueError("The ModelScope snapshot contains no model weights") + hf_files = [ + str(path.relative_to(hf_home)) for path in hf_home.rglob("*") if path.is_file() + ] + # Transformers creates empty import scaffolding even without HF downloads. + empty_scaffolding = {"modules/__init__.py", "modules/hf_remote_code.lock"} + if any( + Path(name).name != "version.txt" + and not ( + name in empty_scaffolding + and not (hf_home / name).is_symlink() + and (hf_home / name).stat().st_size == 0 + ) + for name in hf_files + ): + raise ValueError(f"Unexpected Hugging Face fallback/cache files: {hf_files}") + evidence.update( + snapshot=str(snapshot), assets=assets, hf_files_after_startup=hf_files + ) + report.write_text(json.dumps(evidence, indent=2) + "\n") + print(json.dumps(evidence, indent=2)) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("phase", choices=["before", "after"]) + parser.add_argument("--model", required=True) + parser.add_argument("--cache", type=Path, required=True) + parser.add_argument("--hf-home", type=Path, required=True) + parser.add_argument("--report", type=Path, required=True) + args = parser.parse_args() + operation = record_empty_caches if args.phase == "before" else record_snapshot + operation(args.model, args.cache, args.hf_home, args.report) diff --git a/runners/test_modelscope_snapshot.py b/runners/test_modelscope_snapshot.py new file mode 100644 index 0000000000..503bd3f94c --- /dev/null +++ b/runners/test_modelscope_snapshot.py @@ -0,0 +1,73 @@ +"""Check snapshot provenance with synthetic files and a mocked hub download.""" + +import json +import sys +import types +from pathlib import Path + +import pytest + +from runners.modelscope_snapshot import record_empty_caches, record_snapshot + + +def test_empty_cache_requirement(tmp_path): + cache, hf_home = tmp_path / "modelscope", tmp_path / "hf" + cache.mkdir() + hf_home.mkdir() + report = tmp_path / "report.json" + (cache / "old-file").touch() + with pytest.raises(ValueError, match="Expected an empty cache"): + record_empty_caches("test/model", cache, hf_home, report) + assert not report.exists() + + +def test_snapshot_evidence_and_isolation(tmp_path, monkeypatch): + cache, hf_home = tmp_path / "modelscope", tmp_path / "hf" + cache.mkdir() + hf_home.mkdir() + report = tmp_path / "report.json" + record_empty_caches("test/model", cache, hf_home, report) + snapshot = cache / "snapshot" + snapshot.mkdir() + for name in ( + "config.json", + "tokenizer_config.json", + "tokenizer.json", + "model.safetensors", + ): + (snapshot / name).write_bytes(b"abc") + downloader = types.ModuleType("tensorrt_llm.llmapi.utils") + downloader.download_hf_model = lambda model: snapshot + monkeypatch.setitem(sys.modules, "tensorrt_llm.llmapi.utils", downloader) + + record_snapshot("test/model", cache, hf_home, report) + result = json.loads(report.read_text()) + assert Path(result["snapshot"]).samefile(snapshot) + assert result["modelscope_initial_entries"] == [] + assert result["assets"]["model.safetensors"] == { + "bytes": 3, + "sha256": "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad", + } + assert result["hf_files_after_startup"] == [] + + (hf_home / "modules").mkdir() + for name in ("modules/__init__.py", "modules/hf_remote_code.lock"): + (hf_home / name).touch() + record_snapshot("test/model", cache, hf_home, report) + assert set(json.loads(report.read_text())["hf_files_after_startup"]) == { + "modules/__init__.py", + "modules/hf_remote_code.lock", + } + (hf_home / "modules/__init__.py").write_text("downloaded code") + with pytest.raises(ValueError, match="Hugging Face fallback"): + record_snapshot("test/model", cache, hf_home, report) + (hf_home / "modules/__init__.py").write_text("") + + downloader.download_hf_model = lambda model: tmp_path + with pytest.raises(ValueError, match="outside the fresh ModelScope cache"): + record_snapshot("test/model", cache, hf_home, report) + + downloader.download_hf_model = lambda model: snapshot + (hf_home / "tokenizer.json").touch() + with pytest.raises(ValueError, match="Hugging Face fallback"): + record_snapshot("test/model", cache, hf_home, report) From e78910cb90fd425ef538e244c89e5524da161598 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sun, 20 Sep 2026 19:17:39 -0400 Subject: [PATCH 6/6] docs: link cold-cache evidence to PR 3324 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 PR 3324 更新冷缓存变更记录与引擎补丁豁免链接。 Signed-off-by: functionstackx <47992694+functionstackx@users.noreply.github.com> --- docs/waiver/{3323.md => 3324.md} | 2 +- perf-changelog.yaml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) rename docs/waiver/{3323.md => 3324.md} (98%) diff --git a/docs/waiver/3323.md b/docs/waiver/3324.md similarity index 98% rename from docs/waiver/3323.md rename to docs/waiver/3324.md index e65e1f9829..dc55a2a395 100644 --- a/docs/waiver/3323.md +++ b/docs/waiver/3324.md @@ -1,4 +1,4 @@ -# Inference-engine patch waiver — PR #3323 +# Inference-engine patch waiver — PR #3324 Filed per [`docs/PR_REVIEW_CHECKLIST.md`](../PR_REVIEW_CHECKLIST.md): this PR patches the pinned TensorRT-LLM image before serving because the released image predates ModelScope model loading. diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 70651cdfda..5a7fd75525 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8468,4 +8468,4 @@ description: - "Exercise ModelScope cold downloads in every H100 Qwen3-0.6B job using empty, isolated hub caches and record snapshot hashes before benchmarking or evaluation." - "每个 H100 Qwen3-0.6B 任务均使用独立空缓存测试 ModelScope 冷下载,并在基准或评测前记录模型文件哈希。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3323 + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3324