From 4ea2294be5d59350a6ad80e59f7f500d0bfdf364 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Thu, 1 Oct 2026 20:01:02 +0200 Subject: [PATCH 1/6] Allow workload-specific scenario overrides Signed-off-by: Ivan Podkidyshev --- .../sglang-semantic-eval-override.toml | 28 ++++++++++++ doc/Tutorial.rst | 2 + src/cloudai/models/scenario.py | 43 +++++++++++-------- tests/test_test_scenario.py | 29 +++++++++++++ 4 files changed, 84 insertions(+), 18 deletions(-) create mode 100644 conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml diff --git a/conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml b/conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml new file mode 100644 index 000000000..1452d4c82 --- /dev/null +++ b/conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml @@ -0,0 +1,28 @@ +# SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES +# Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Evaluate 20 examples instead of 200 while inheriting the entrypoint and all +# other test settings from the referenced SGLang test. +name = "sglang-semantic-eval-override" + +[[Tests]] +id = "sglang.semantic-eval.20-examples" +path = "../test/sglang.toml" +num_nodes = 1 +time_limit = "00:10:00" + + [Tests.semantic_eval_cmd_args] + cli = "--host {host} --port {port} --eval-name gsm8k --num-examples 20 --num-threads 128 --model {model}" diff --git a/doc/Tutorial.rst b/doc/Tutorial.rst index 47ab92769..01f13232c 100644 --- a/doc/Tutorial.rst +++ b/doc/Tutorial.rst @@ -278,5 +278,7 @@ It is possible to override some args or even fully define a workload inside a sc ``allreduce.in.scenario`` fully defines a workload; in this case ``test_name`` must not be set, while ``name``, ``description`` and ``test_template_name`` must be set. ``allreduce.override`` overrides only ``stepfactor`` arg from the test defined in the tests directory. +Scenario entries can also override workload-specific fields, such as ``semantic_eval_cmd_args`` in an SGLang test. +The selected workload validates these fields after they are merged with the referenced test. If a scenario contains only fully defined tests, ``--tests-dir`` arg is not required. diff --git a/src/cloudai/models/scenario.py b/src/cloudai/models/scenario.py index 25d61527f..543a0cf3c 100644 --- a/src/cloudai/models/scenario.py +++ b/src/cloudai/models/scenario.py @@ -63,12 +63,30 @@ class TestRunDependencyModel(BaseModel): id: str +SCENARIO_TEST_RUN_FIELDS: set[str] = { + "id", + "test_name", + "path", + "num_nodes", + "nodes", + "pin_nodes", + "exclude_nodes", + "weight", + "iterations", + "sol", + "ideal_perf", + "time_limit", + "dependencies", + "extra_srun_args", +} + + class TestRunModel(BaseModel): """Model for test run in test scenario.""" __test__ = False - model_config = ConfigDict(extra="forbid") + model_config = ConfigDict(extra="allow") id: str = Field(min_length=1) test_name: Optional[str] = None @@ -128,23 +146,12 @@ def validate_pin_nodes(self) -> Self: return self def tdef_model_dump(self, by_alias: bool) -> dict: - """Return a dictionary with non-None values that correspond to the test definition fields.""" - data = { - "name": self.name, - "description": self.description, - "test_template_name": self.test_template_name, - "agent": self.agent, - "agent_steps": self.agent_steps, - "agent_metrics": self.agent_metrics if "agent_metrics" in self.model_fields_set else None, - "agent_reward_function": self.agent_reward_function, - "agent_config": self.agent_config, - "dse_excluded_args": self.dse_excluded_args, - "extra_container_mounts": self.extra_container_mounts, - "extra_env_vars": self.extra_env_vars if self.extra_env_vars else None, - "cmd_args": self.cmd_args.model_dump(by_alias=by_alias) if self.cmd_args else None, - "git_repos": [repo.model_dump() for repo in self.git_repos] if self.git_repos else None, - "nsys": self.nsys.model_dump(exclude_unset=True) if self.nsys else None, - } + """Return explicitly set workload fields, including workload-specific fields.""" + data = self.model_dump(by_alias=by_alias, exclude=SCENARIO_TEST_RUN_FIELDS, exclude_unset=True) + # Empty values for these fields did not replace values in referenced tests. + for field in ("extra_env_vars", "git_repos"): + if not data.get(field): + data.pop(field, None) return {k: v for k, v in data.items() if v is not None} @model_validator(mode="after") diff --git a/tests/test_test_scenario.py b/tests/test_test_scenario.py index cefb462f0..60f336d43 100644 --- a/tests/test_test_scenario.py +++ b/tests/test_test_scenario.py @@ -38,6 +38,7 @@ ) from cloudai.models.scenario import TestRunModel, TestScenarioModel from cloudai.systems.slurm.slurm_system import SlurmSystem +from cloudai.test_parser import TestParser from cloudai.test_scenario_parser import calculate_total_time_limit, get_reporters from cloudai.workloads.ai_dynamo import AIDynamoReportGenerationStrategy, AIDynamoTestDefinition from cloudai.workloads.aiconfig import AiconfiguratorReportGenerationStrategy, AiconfiguratorTestDefinition @@ -314,6 +315,34 @@ def test_resumes_from_pre_existing_value(self, tdef: TestDefinition) -> None: class TestInScenario: + @pytest.mark.parametrize("reference", ["path", "test_name"]) + def test_workload_specific_override(self, reference: str, slurm_system: SlurmSystem): + scenario_path = ( + Path(__file__).resolve().parents[1] + / "conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml" + ) + scenario_data = toml.load(scenario_path) + test_path = scenario_path.parent / scenario_data["Tests"][0]["path"] + base_test = TestParser([test_path], slurm_system).parse_all()[0] + + if reference == "test_name": + scenario_data["Tests"][0].pop("path") + scenario_data["Tests"][0]["test_name"] = base_test.name + + parser = TestScenarioParser(scenario_path, slurm_system, {base_test.name: base_test}, {}) + test = parser._parse_data(scenario_data).test_runs[0].test + + assert isinstance(test, SglangTestDefinition) + assert isinstance(base_test, SglangTestDefinition) + assert test.semantic_eval_cmd_args is not None + assert base_test.semantic_eval_cmd_args is not None + assert test.semantic_eval_cmd_args.cli == scenario_data["Tests"][0]["semantic_eval_cmd_args"]["cli"] + assert test.semantic_eval_cmd_args.entrypoint == base_test.semantic_eval_cmd_args.entrypoint + + scenario_data["Tests"][0]["unknown_workload_field"] = 1 + with pytest.raises(TestConfigParsingError): + parser._parse_data(scenario_data) + @pytest.mark.parametrize("missing_arg", ["test_template_name", "name", "description"]) def test_without_base(self, missing_arg: str): spec = { From 29b4cca69fde8a8d88efd160500dc2415866642c Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Thu, 1 Oct 2026 20:08:01 +0200 Subject: [PATCH 2/6] Keep scenario override example in parser test Signed-off-by: Ivan Podkidyshev --- .../sglang-semantic-eval-override.toml | 28 ------------------- tests/test_test_scenario.py | 16 ++++++++--- 2 files changed, 12 insertions(+), 32 deletions(-) delete mode 100644 conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml diff --git a/conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml b/conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml deleted file mode 100644 index 1452d4c82..000000000 --- a/conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml +++ /dev/null @@ -1,28 +0,0 @@ -# SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES -# Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# Evaluate 20 examples instead of 200 while inheriting the entrypoint and all -# other test settings from the referenced SGLang test. -name = "sglang-semantic-eval-override" - -[[Tests]] -id = "sglang.semantic-eval.20-examples" -path = "../test/sglang.toml" -num_nodes = 1 -time_limit = "00:10:00" - - [Tests.semantic_eval_cmd_args] - cli = "--host {host} --port {port} --eval-name gsm8k --num-examples 20 --num-threads 128 --model {model}" diff --git a/tests/test_test_scenario.py b/tests/test_test_scenario.py index 60f336d43..eba215aaa 100644 --- a/tests/test_test_scenario.py +++ b/tests/test_test_scenario.py @@ -317,11 +317,19 @@ def test_resumes_from_pre_existing_value(self, tdef: TestDefinition) -> None: class TestInScenario: @pytest.mark.parametrize("reference", ["path", "test_name"]) def test_workload_specific_override(self, reference: str, slurm_system: SlurmSystem): - scenario_path = ( - Path(__file__).resolve().parents[1] - / "conf/experimental/sglang/test_scenario/sglang-semantic-eval-override.toml" + scenario_path = Path(__file__).resolve().parents[1] / "conf/experimental/sglang/test_scenario/sglang.toml" + scenario_data = toml.loads( + """ + name = "sglang-override" + + [[Tests]] + id = "semantic-eval" + path = "../test/sglang.toml" + + [Tests.semantic_eval_cmd_args] + cli = "--host {host} --port {port} --eval-name gsm8k --num-examples 20 --num-threads 128 --model {model}" + """ ) - scenario_data = toml.load(scenario_path) test_path = scenario_path.parent / scenario_data["Tests"][0]["path"] base_test = TestParser([test_path], slurm_system).parse_all()[0] From 758e9be8802370a5181f9624232e0d80bd25c536 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Thu, 1 Oct 2026 20:19:03 +0200 Subject: [PATCH 3/6] Document scenario workload overrides Signed-off-by: Ivan Podkidyshev --- doc/Tutorial.rst | 3 +-- doc/workloads/sglang.rst | 5 ++--- doc/workloads/vllm.rst | 5 ++--- 3 files changed, 5 insertions(+), 8 deletions(-) diff --git a/doc/Tutorial.rst b/doc/Tutorial.rst index 01f13232c..f635c9299 100644 --- a/doc/Tutorial.rst +++ b/doc/Tutorial.rst @@ -192,6 +192,7 @@ Notes on the test scenario: #. ``id`` is a mandatory field and must be unique for each test. #. The ``test_name`` specifies the test definition from one of the Test TOML files. Node lists and time limits are optional. +#. All workload definition parameters can be overridden in a test case, including workload-specific sections such as ``semantic_eval_cmd_args``. The referenced test's ``test_template_name`` cannot be changed. #. If needed, ``nodes`` should be described as a list of node names as shown in a Slurm system. Alternatively, if groups are defined in the system schema, you can ask CloudAI to allocate a specific number of nodes from a specified partition and group. For example, ``nodes = ['PARTITION:GROUP:16']`` allocates 16 nodes from group ``GROUP`` and partition ``PARTITION``. #. There are three types of dependencies: ``start_post_comp``, ``start_post_init`` and ``end_post_comp``. @@ -278,7 +279,5 @@ It is possible to override some args or even fully define a workload inside a sc ``allreduce.in.scenario`` fully defines a workload; in this case ``test_name`` must not be set, while ``name``, ``description`` and ``test_template_name`` must be set. ``allreduce.override`` overrides only ``stepfactor`` arg from the test defined in the tests directory. -Scenario entries can also override workload-specific fields, such as ``semantic_eval_cmd_args`` in an SGLang test. -The selected workload validates these fields after they are merged with the referenced test. If a scenario contains only fully defined tests, ``--tests-dir`` arg is not required. diff --git a/doc/workloads/sglang.rst b/doc/workloads/sglang.rst index 237537af4..78e70e303 100644 --- a/doc/workloads/sglang.rst +++ b/doc/workloads/sglang.rst @@ -65,9 +65,8 @@ Test-in-Scenario example docker_image_url = "lmsysorg/sglang:dev-cu13" model = "Qwen/Qwen3-8B" -Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, are not -supported under ``[[Tests]]`` in a test scenario. Define them in a test definition TOML and reference that test with -``test_name`` when custom benchmark or semantic-evaluation arguments are needed. +Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, can be set or +overridden under ``[[Tests]]`` in a test scenario; see :ref:`test-in-scenario`. Local Models diff --git a/doc/workloads/vllm.rst b/doc/workloads/vllm.rst index 082faf70d..fb2a9a634 100644 --- a/doc/workloads/vllm.rst +++ b/doc/workloads/vllm.rst @@ -65,9 +65,8 @@ Test-in-Scenario example docker_image_url = "nvcr.io#nvidia/ai-dynamo/vllm-runtime:0.7.0" model = "Qwen/Qwen3-0.6B" -Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, are not -supported under ``[[Tests]]`` in a test scenario. Define them in a test definition TOML and reference that test with -``test_name`` when custom benchmark or semantic-evaluation arguments are needed. +Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, can be set or +overridden under ``[[Tests]]`` in a test scenario; see :ref:`test-in-scenario`. Local Models From a3afad1a748b85410ae398ddc30ad9d6512e88ac Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Thu, 1 Oct 2026 20:21:42 +0200 Subject: [PATCH 4/6] Remove duplicate workload override notes Signed-off-by: Ivan Podkidyshev --- doc/workloads/sglang.rst | 4 ---- doc/workloads/vllm.rst | 4 ---- 2 files changed, 8 deletions(-) diff --git a/doc/workloads/sglang.rst b/doc/workloads/sglang.rst index 78e70e303..5936800c5 100644 --- a/doc/workloads/sglang.rst +++ b/doc/workloads/sglang.rst @@ -65,10 +65,6 @@ Test-in-Scenario example docker_image_url = "lmsysorg/sglang:dev-cu13" model = "Qwen/Qwen3-8B" -Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, can be set or -overridden under ``[[Tests]]`` in a test scenario; see :ref:`test-in-scenario`. - - Local Models ------------ Set ``cmd_args.model`` to an absolute, container-visible path to load an existing model directory instead of diff --git a/doc/workloads/vllm.rst b/doc/workloads/vllm.rst index fb2a9a634..745794dc9 100644 --- a/doc/workloads/vllm.rst +++ b/doc/workloads/vllm.rst @@ -65,10 +65,6 @@ Test-in-Scenario example docker_image_url = "nvcr.io#nvidia/ai-dynamo/vllm-runtime:0.7.0" model = "Qwen/Qwen3-0.6B" -Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, can be set or -overridden under ``[[Tests]]`` in a test scenario; see :ref:`test-in-scenario`. - - Local Models ------------ Set ``cmd_args.model`` to an absolute, container-visible path to load an existing model directory instead of From 01582d96e5306f78b86280db138a296688f06274 Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Thu, 1 Oct 2026 20:33:19 +0200 Subject: [PATCH 5/6] Consolidate NeMo Launcher FP8 two-node config Signed-off-by: Ivan Podkidyshev --- ...nemo_launcher_nemotron_15b_fp8_2_node.toml | 57 ------------------- .../nemo_launcher_nemotron_15b_fp8.toml | 10 +++- ...nemo_launcher_nemotron_15b_fp8_2_node.toml | 10 +++- 3 files changed, 18 insertions(+), 59 deletions(-) delete mode 100644 conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml diff --git a/conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml b/conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml deleted file mode 100644 index bdf700972..000000000 --- a/conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml +++ /dev/null @@ -1,57 +0,0 @@ -# SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES -# Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -name = "nemo_launcher_nemotron_15b_fp8_2_node" -description = "nemo_launcher_nemotron_15b_fp8_2_node" -test_template_name = "NeMoLauncher" - -[cmd_args] -docker_image_url = "nvcr.io#nvidia/nemo:25.07" - [cmd_args.training] - values = "nemotron/nemotron_15b" - - [cmd_args.training.run] - time_limit = "00:45:00" - - [cmd_args.training.trainer] - limit_val_batches = "1" - max_steps = "100" - precision = "bf16" - val_check_interval = "100" - - [cmd_args.training.model] - bias_activation_fusion = "false" - fp8 = "True" - fp8_hybrid = "True" - global_batch_size = 64 - mcore_gpt = "True" - micro_batch_size = 4 - pipeline_model_parallel_size = 1 - sequence_parallel = "True" - tensor_model_parallel_size = 4 - transformer_engine = "True" - ub_tp_comm_overlap = "True" - virtual_pipeline_model_parallel_size = "null" - - [cmd_args.training.model.tokenizer] - model = "/path/to/nemotron_2_256k.model" - - [cmd_args.training.exp_manager] - create_wandb_logger = "false" - -[extra_cmd_args] -"env_vars.NVTE_FUSED_ATTN" = "1" -"+training.model.fp8_params" = "True" diff --git a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml index 90911c942..5b6c871a5 100644 --- a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml +++ b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml @@ -18,9 +18,17 @@ name = "nemo_launcher_nemotron_15b_fp8" [[Tests]] id = "nemo_launcher_nemotron_15b_fp8_2_node" -test_name = "nemo_launcher_nemotron_15b_fp8_2_node" +path = "../test/nemo_launcher_nemotron_15b_bf16_2_node.toml" +name = "nemo_launcher_nemotron_15b_fp8_2_node" +description = "nemo_launcher_nemotron_15b_fp8_2_node" num_nodes = "2" + [Tests.cmd_args.training.model] + fp8 = "True" + + [Tests.extra_cmd_args] + "+training.model.fp8_params" = "True" + [[Tests]] id = "nemo_launcher_nemotron_15b_fp8_4_node" test_name = "nemo_launcher_nemotron_15b_fp8_4_node" diff --git a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml index e77f19410..d55e6bf3d 100644 --- a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml +++ b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml @@ -18,5 +18,13 @@ name = "nemo_launcher_nemotron_15b_fp8_2_node" [[Tests]] id = "nemo_launcher_nemotron_15b_fp8_2_node" -test_name = "nemo_launcher_nemotron_15b_fp8_2_node" +path = "../test/nemo_launcher_nemotron_15b_bf16_2_node.toml" +name = "nemo_launcher_nemotron_15b_fp8_2_node" +description = "nemo_launcher_nemotron_15b_fp8_2_node" num_nodes = "2" + + [Tests.cmd_args.training.model] + fp8 = "True" + + [Tests.extra_cmd_args] + "+training.model.fp8_params" = "True" From 3519c88556726b3095dd07617597c49ea92c1cfb Mon Sep 17 00:00:00 2001 From: Ivan Podkidyshev Date: Thu, 1 Oct 2026 20:49:19 +0200 Subject: [PATCH 6/6] Update copyright year in NeMo scenario Signed-off-by: Ivan Podkidyshev --- .../test_scenario/nemo_launcher_nemotron_15b_fp8.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml index 5b6c871a5..c8f981435 100644 --- a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml +++ b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml @@ -1,5 +1,5 @@ # SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES -# Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # # Licensed under the Apache License, Version 2.0 (the "License");