diff --git a/conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml b/conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml deleted file mode 100644 index bdf700972..000000000 --- a/conf/experimental/test/nemo_launcher_nemotron_15b_fp8_2_node.toml +++ /dev/null @@ -1,57 +0,0 @@ -# SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES -# Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -name = "nemo_launcher_nemotron_15b_fp8_2_node" -description = "nemo_launcher_nemotron_15b_fp8_2_node" -test_template_name = "NeMoLauncher" - -[cmd_args] -docker_image_url = "nvcr.io#nvidia/nemo:25.07" - [cmd_args.training] - values = "nemotron/nemotron_15b" - - [cmd_args.training.run] - time_limit = "00:45:00" - - [cmd_args.training.trainer] - limit_val_batches = "1" - max_steps = "100" - precision = "bf16" - val_check_interval = "100" - - [cmd_args.training.model] - bias_activation_fusion = "false" - fp8 = "True" - fp8_hybrid = "True" - global_batch_size = 64 - mcore_gpt = "True" - micro_batch_size = 4 - pipeline_model_parallel_size = 1 - sequence_parallel = "True" - tensor_model_parallel_size = 4 - transformer_engine = "True" - ub_tp_comm_overlap = "True" - virtual_pipeline_model_parallel_size = "null" - - [cmd_args.training.model.tokenizer] - model = "/path/to/nemotron_2_256k.model" - - [cmd_args.training.exp_manager] - create_wandb_logger = "false" - -[extra_cmd_args] -"env_vars.NVTE_FUSED_ATTN" = "1" -"+training.model.fp8_params" = "True" diff --git a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml index 90911c942..c8f981435 100644 --- a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml +++ b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8.toml @@ -1,5 +1,5 @@ # SPDX-FileCopyrightText: NVIDIA CORPORATION & AFFILIATES -# Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # # Licensed under the Apache License, Version 2.0 (the "License"); @@ -18,9 +18,17 @@ name = "nemo_launcher_nemotron_15b_fp8" [[Tests]] id = "nemo_launcher_nemotron_15b_fp8_2_node" -test_name = "nemo_launcher_nemotron_15b_fp8_2_node" +path = "../test/nemo_launcher_nemotron_15b_bf16_2_node.toml" +name = "nemo_launcher_nemotron_15b_fp8_2_node" +description = "nemo_launcher_nemotron_15b_fp8_2_node" num_nodes = "2" + [Tests.cmd_args.training.model] + fp8 = "True" + + [Tests.extra_cmd_args] + "+training.model.fp8_params" = "True" + [[Tests]] id = "nemo_launcher_nemotron_15b_fp8_4_node" test_name = "nemo_launcher_nemotron_15b_fp8_4_node" diff --git a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml index e77f19410..d55e6bf3d 100644 --- a/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml +++ b/conf/experimental/test_scenario/nemo_launcher_nemotron_15b_fp8_2_node.toml @@ -18,5 +18,13 @@ name = "nemo_launcher_nemotron_15b_fp8_2_node" [[Tests]] id = "nemo_launcher_nemotron_15b_fp8_2_node" -test_name = "nemo_launcher_nemotron_15b_fp8_2_node" +path = "../test/nemo_launcher_nemotron_15b_bf16_2_node.toml" +name = "nemo_launcher_nemotron_15b_fp8_2_node" +description = "nemo_launcher_nemotron_15b_fp8_2_node" num_nodes = "2" + + [Tests.cmd_args.training.model] + fp8 = "True" + + [Tests.extra_cmd_args] + "+training.model.fp8_params" = "True" diff --git a/doc/Tutorial.rst b/doc/Tutorial.rst index 47ab92769..f635c9299 100644 --- a/doc/Tutorial.rst +++ b/doc/Tutorial.rst @@ -192,6 +192,7 @@ Notes on the test scenario: #. ``id`` is a mandatory field and must be unique for each test. #. The ``test_name`` specifies the test definition from one of the Test TOML files. Node lists and time limits are optional. +#. All workload definition parameters can be overridden in a test case, including workload-specific sections such as ``semantic_eval_cmd_args``. The referenced test's ``test_template_name`` cannot be changed. #. If needed, ``nodes`` should be described as a list of node names as shown in a Slurm system. Alternatively, if groups are defined in the system schema, you can ask CloudAI to allocate a specific number of nodes from a specified partition and group. For example, ``nodes = ['PARTITION:GROUP:16']`` allocates 16 nodes from group ``GROUP`` and partition ``PARTITION``. #. There are three types of dependencies: ``start_post_comp``, ``start_post_init`` and ``end_post_comp``. diff --git a/doc/workloads/sglang.rst b/doc/workloads/sglang.rst index 237537af4..5936800c5 100644 --- a/doc/workloads/sglang.rst +++ b/doc/workloads/sglang.rst @@ -65,11 +65,6 @@ Test-in-Scenario example docker_image_url = "lmsysorg/sglang:dev-cu13" model = "Qwen/Qwen3-8B" -Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, are not -supported under ``[[Tests]]`` in a test scenario. Define them in a test definition TOML and reference that test with -``test_name`` when custom benchmark or semantic-evaluation arguments are needed. - - Local Models ------------ Set ``cmd_args.model`` to an absolute, container-visible path to load an existing model directory instead of diff --git a/doc/workloads/vllm.rst b/doc/workloads/vllm.rst index 082faf70d..745794dc9 100644 --- a/doc/workloads/vllm.rst +++ b/doc/workloads/vllm.rst @@ -65,11 +65,6 @@ Test-in-Scenario example docker_image_url = "nvcr.io#nvidia/ai-dynamo/vllm-runtime:0.7.0" model = "Qwen/Qwen3-0.6B" -Workload-specific test definition sections, such as ``bench_cmd_args`` and ``semantic_eval_cmd_args``, are not -supported under ``[[Tests]]`` in a test scenario. Define them in a test definition TOML and reference that test with -``test_name`` when custom benchmark or semantic-evaluation arguments are needed. - - Local Models ------------ Set ``cmd_args.model`` to an absolute, container-visible path to load an existing model directory instead of diff --git a/src/cloudai/models/scenario.py b/src/cloudai/models/scenario.py index 25d61527f..543a0cf3c 100644 --- a/src/cloudai/models/scenario.py +++ b/src/cloudai/models/scenario.py @@ -63,12 +63,30 @@ class TestRunDependencyModel(BaseModel): id: str +SCENARIO_TEST_RUN_FIELDS: set[str] = { + "id", + "test_name", + "path", + "num_nodes", + "nodes", + "pin_nodes", + "exclude_nodes", + "weight", + "iterations", + "sol", + "ideal_perf", + "time_limit", + "dependencies", + "extra_srun_args", +} + + class TestRunModel(BaseModel): """Model for test run in test scenario.""" __test__ = False - model_config = ConfigDict(extra="forbid") + model_config = ConfigDict(extra="allow") id: str = Field(min_length=1) test_name: Optional[str] = None @@ -128,23 +146,12 @@ def validate_pin_nodes(self) -> Self: return self def tdef_model_dump(self, by_alias: bool) -> dict: - """Return a dictionary with non-None values that correspond to the test definition fields.""" - data = { - "name": self.name, - "description": self.description, - "test_template_name": self.test_template_name, - "agent": self.agent, - "agent_steps": self.agent_steps, - "agent_metrics": self.agent_metrics if "agent_metrics" in self.model_fields_set else None, - "agent_reward_function": self.agent_reward_function, - "agent_config": self.agent_config, - "dse_excluded_args": self.dse_excluded_args, - "extra_container_mounts": self.extra_container_mounts, - "extra_env_vars": self.extra_env_vars if self.extra_env_vars else None, - "cmd_args": self.cmd_args.model_dump(by_alias=by_alias) if self.cmd_args else None, - "git_repos": [repo.model_dump() for repo in self.git_repos] if self.git_repos else None, - "nsys": self.nsys.model_dump(exclude_unset=True) if self.nsys else None, - } + """Return explicitly set workload fields, including workload-specific fields.""" + data = self.model_dump(by_alias=by_alias, exclude=SCENARIO_TEST_RUN_FIELDS, exclude_unset=True) + # Empty values for these fields did not replace values in referenced tests. + for field in ("extra_env_vars", "git_repos"): + if not data.get(field): + data.pop(field, None) return {k: v for k, v in data.items() if v is not None} @model_validator(mode="after") diff --git a/tests/test_test_scenario.py b/tests/test_test_scenario.py index cefb462f0..eba215aaa 100644 --- a/tests/test_test_scenario.py +++ b/tests/test_test_scenario.py @@ -38,6 +38,7 @@ ) from cloudai.models.scenario import TestRunModel, TestScenarioModel from cloudai.systems.slurm.slurm_system import SlurmSystem +from cloudai.test_parser import TestParser from cloudai.test_scenario_parser import calculate_total_time_limit, get_reporters from cloudai.workloads.ai_dynamo import AIDynamoReportGenerationStrategy, AIDynamoTestDefinition from cloudai.workloads.aiconfig import AiconfiguratorReportGenerationStrategy, AiconfiguratorTestDefinition @@ -314,6 +315,42 @@ def test_resumes_from_pre_existing_value(self, tdef: TestDefinition) -> None: class TestInScenario: + @pytest.mark.parametrize("reference", ["path", "test_name"]) + def test_workload_specific_override(self, reference: str, slurm_system: SlurmSystem): + scenario_path = Path(__file__).resolve().parents[1] / "conf/experimental/sglang/test_scenario/sglang.toml" + scenario_data = toml.loads( + """ + name = "sglang-override" + + [[Tests]] + id = "semantic-eval" + path = "../test/sglang.toml" + + [Tests.semantic_eval_cmd_args] + cli = "--host {host} --port {port} --eval-name gsm8k --num-examples 20 --num-threads 128 --model {model}" + """ + ) + test_path = scenario_path.parent / scenario_data["Tests"][0]["path"] + base_test = TestParser([test_path], slurm_system).parse_all()[0] + + if reference == "test_name": + scenario_data["Tests"][0].pop("path") + scenario_data["Tests"][0]["test_name"] = base_test.name + + parser = TestScenarioParser(scenario_path, slurm_system, {base_test.name: base_test}, {}) + test = parser._parse_data(scenario_data).test_runs[0].test + + assert isinstance(test, SglangTestDefinition) + assert isinstance(base_test, SglangTestDefinition) + assert test.semantic_eval_cmd_args is not None + assert base_test.semantic_eval_cmd_args is not None + assert test.semantic_eval_cmd_args.cli == scenario_data["Tests"][0]["semantic_eval_cmd_args"]["cli"] + assert test.semantic_eval_cmd_args.entrypoint == base_test.semantic_eval_cmd_args.entrypoint + + scenario_data["Tests"][0]["unknown_workload_field"] = 1 + with pytest.raises(TestConfigParsingError): + parser._parse_data(scenario_data) + @pytest.mark.parametrize("missing_arg", ["test_template_name", "name", "description"]) def test_without_base(self, missing_arg: str): spec = {