Files
Model-Optimizer/tools/launcher/tests/test_yaml_formats.py
T
Chenhan D. YuandClaude Opus 4.6 839fa3d658 add: ModelOpt Launcher for Slurm job submission (#1031)
```
# Install                                                                                                                                                                              
cd Model-Optimizer/launcher                                                                                                                                                            
curl -LsSf https://astral.sh/uv/install.sh | sh                                                                                                                                        
git submodule update --init --recursive                                                                                                                                                
                                                                                                                                                                                         
# Run locally with Docker (single GPU)                                                                                                                                                 
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml hf_local=/mnt/hf-local --yes                                                                                              
                                                                                                                                                                                         
# Run on Slurm cluster (no need to export the follow SLURM_XXX envs if used in sandbox)                                                                                                                                                              
export SLURM_HOST=login-node.example.com                                                                                                                                               
export SLURM_ACCOUNT=my_account                                                                                                                                                        
export SLURM_HF_LOCAL=/shared/hf-local                                                                                                                                                 
export SLURM_JOB_DIR=/shared/experiments                                                                                                                                               
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --yes                                                                                                                       
                                                                                                                                                                                         
# Preview config without running                                                                                                                                                       
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --dryrun --yes -v                                                                                                           
                                                                                                                                                                                         
# Override parameters                                                                                                                                                                  
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml \                                                                                                                           
      pipeline.task_0.slurm_config.nodes=2 --yes                                                                                                                                         
                                                                                                                                                                                         
# Dump resolved config for reproducibility (single YAML for reproducibility, great for QA, Eng, and agent to triage)
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --to-yaml resolved.yaml                                                                                                     
                                                                                                                                                                                         
# Run tests
uv pip install -e . pytest                                                                                                                                                             
uv run pytest -v 
```
 ## Summary

Add `launcher/` module for submitting quantization, training, and
evaluation jobs to Slurm clusters or running them locally with Docker
via `nemo-run`. `nemo-run` is used in all `NVIDIA-NeMo/*` projects. It
supports modern YAML factory (superset of the `OmegaConf` and `Hydra`)
and it support multiple executor backends (here we use docker and slurm
mainly).

A sample YAML config `launcher/Qwen/Qwen3-8B/megatron_lm_ptq.yaml`:
```
job_name: Qwen3-8B_NVFP4_DEFAULT_CFG
pipeline:
  # hf_local: path prefix for model weights and datasets.
  #
  # This should be a self-managed directory that mirrors the HuggingFace Hub
  # hierarchy (e.g., /hf-local/Qwen/Qwen3-8B/, /hf-local/cais/mmlu/). Using
  # a dedicated folder is preferred over the HuggingFace cache (~/.cache/huggingface)
  # to avoid cache corruption issues with concurrent jobs.
  #
  # Override on CLI:
  #   pipeline.global_vars.hf_local=/mnt/my-models/   # use a different path
  #   pipeline.global_vars.hf_local=""                 # download from HuggingFace Hub
  global_vars:
    hf_local: /hf-local/

  task_0:
    script: common/megatron-lm/quantize/quantize.sh
    args:
      - --calib-dataset-path-or-name <<global_vars.hf_local>>abisee/cnn_dailymail
      - --calib-size 32
    environment:
      - MLM_MODEL_CFG: Qwen/Qwen3-8B
      - QUANT_CFG: NVFP4_DEFAULT_CFG
      - HF_MODEL_CKPT: <<global_vars.hf_local>>Qwen/Qwen3-8B
      - MMLU_DATASET: <<global_vars.hf_local>>cais/mmlu
      - TP: 4
    slurm_config:
      _factory_: "slurm_factory"
      nodes: 1
      ntasks_per_node: 4
      gpus_per_node: 4
```
  
### Key features
  - **`launch.py`** — public entrypoint accepting `--yaml` config format
- **`core.py`** — shared logic (dataclasses, executor builders, run
loop) also used by nmm-sandbox's `slurm.py`
- **Factory system** — env-var-driven `slurm_factory` with
`register_factory()` registry
- **`<<global_vars.X>>`** interpolation for sharing values across
pipeline tasks
- **`hf_local`** global var for configurable model/dataset storage path
- **Version reporting** — git commit/branch printed at job start for
reproducibility
- **`--to-yaml`** — dump resolved config for bug reports and
reproducibility
- **Model-Optimizer symlink** — `modules/Model-Optimizer -> ../..`
(auto-created, avoids recursive submodule)
  ### Files
| Path | Description |
  |------|-------------|
  | `launcher/launch.py` | Public entrypoint |
  | `launcher/core.py` | Shared dataclasses, executors, run loop |
| `launcher/slurm_config.py` | SlurmConfig + env-var factory |
| `launcher/common/` | Shell scripts (quantize, query, eagle3,
specdec_bench) |
| `launcher/Qwen/Qwen3-8B/` | Example configs (PTQ, EAGLE3 pipeline) |
| `launcher/tests/` | 64 unit tests |
  | `launcher/README.md` | User guide |
| `launcher/ADVANCED.md` | Architecture, mount mechanism, Claude Code
workflows |
| `launcher/CLAUDE.md` | Claude Code project instructions |
  | `.github/workflows/unit_tests.yml` | CI job for launcher tests |
### Verified

- Same YAML produces identical MMLU results via both `slurm.py` and
`launch.py`:
    - Local Docker (TP=1): 0.719 (128/178)
- OCI-HSG Slurm (TP=4): 0.730 (130/178)
  ## Test plan
- [x] 64 unit tests (core, factory, YAML, Docker executor, Slurm
executor, Docker launch)
  - [x] CI workflow added to `.github/workflows/unit_tests.yml`
- [x] Local Docker end-to-end with `python:3.12-slim`
- [x] Qwen3-8B PTQ on OCI-HSG via both launchers
- [ ] Reviewer runs: `cd launcher && uv pip install -e . pytest && uv
run pytest -v`

### Before your PR is "*Ready for review*"

Make sure you read and follow [Contributor
guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md)
and your commits are signed (`git commit -s -S`).

Make sure you read and follow the [Security Best
Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors)
(e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(...,
weights_only=False)`, `pickle`, etc.).

- Is this change backward compatible?: ✅ / ❌ / N/A <!--- If ❌, explain
why. -->
- If you copied code from any other sources or added a new PIP
dependency, did you follow guidance in `CONTRIBUTING.md`: ✅ / ❌ / N/A
<!--- Mandatory -->
- Did you write any new necessary tests?: ✅ / ❌ / N/A <!--- Mandatory
for new features or examples. -->
- Did you update
[Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?:
✅ / ❌ / N/A <!--- Only for new features, API changes, critical bug fixes
or backward incompatible changes. -->

### Additional Information
<!-- E.g. related issue. -->


<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

## Release Notes

* **New Features**
* Introduced ModelOpt Launcher for submitting quantization, training,
and evaluation jobs to Slurm clusters or running locally via Docker.
* Added YAML-based job configuration with multi-task pipeline support
and global variable interpolation.
* Included example workflows for Qwen3-8B quantization and EAGLE3
speculative decoding.
  * Provided configurable Slurm and execution environment defaults.

* **Documentation**
* Added comprehensive README with quick start, environment setup, and
configuration guidance.
* Added advanced guide detailing launcher architecture and integration
patterns.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->

---------

Signed-off-by: Chenhan Yu <chenhany@nvidia.com>
Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-18 15:32:15 -07:00

193 lines
6.4 KiB
Python

# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""Tests for YAML config parsing — verifies that different YAML formats produce correct dataclasses.
Coverage:
- --yaml format: top-level job_name + pipeline with task_0, environment, slurm_config
- pipeline=@ format: bare SandboxPipeline without job_name wrapper
- task_configs: list of YAML paths resolved via factory registry
- Environment formats: list-of-dicts and flat dict both parsed correctly
- Global vars: <<global_vars.X>> resolved in both args and environment
"""
import yaml
class TestYamlFormatParsing:
"""Tests that YAML content parses into correct dataclass structures."""
def test_yaml_format_with_job_name(self, tmp_yaml):
"""The --yaml format has job_name and pipeline as top-level keys."""
content = """
job_name: test_job
pipeline:
skip: false
allow_to_fail: true
note: "test note"
task_0:
script: test.sh
args:
- --flag
environment:
- KEY: value
"""
path = tmp_yaml(content)
with open(path) as f:
data = yaml.safe_load(f)
assert data["job_name"] == "test_job"
assert data["pipeline"]["skip"] is False
assert data["pipeline"]["allow_to_fail"] is True
assert data["pipeline"]["note"] == "test note"
assert data["pipeline"]["task_0"]["script"] == "test.sh"
assert data["pipeline"]["task_0"]["args"] == ["--flag"]
assert data["pipeline"]["task_0"]["environment"] == [{"KEY": "value"}]
def test_bare_pipeline_format(self, tmp_yaml):
"""The pipeline=@ format is a bare SandboxPipeline without wrapper."""
content = """
task_0:
script: a.sh
args:
- --foo
task_1:
script: b.sh
allow_to_fail: false
skip: false
"""
path = tmp_yaml(content)
with open(path) as f:
data = yaml.safe_load(f)
# Verify the YAML parses into valid SandboxPipeline kwargs
# (nemo-run does this via its CLI parser; we just verify the structure)
assert "task_0" in data
assert "task_1" in data
assert data["task_0"]["script"] == "a.sh"
assert data["task_1"]["script"] == "b.sh"
def test_task_configs_format(self, tmp_yaml):
"""task_configs lists YAML files that are resolved into tasks."""
from core import SandboxPipeline, register_factory
def local_factory(nodes=1):
return {"nodes": nodes}
register_factory("local_factory", local_factory)
task_path = tmp_yaml(
"""
script: worker.sh
args:
- --batch-size 32
slurm_config:
_factory_: "local_factory"
nodes: 2
""",
name="worker.yaml",
)
pipeline = SandboxPipeline(task_configs=[task_path])
assert len(pipeline.tasks) == 1
assert pipeline.tasks[0].script == "worker.sh"
assert pipeline.tasks[0].args == ["--batch-size 32"]
assert pipeline.tasks[0].slurm_config == {"nodes": 2}
def test_environment_list_of_dicts(self):
"""Environment as list-of-single-key-dicts (nemo-run format)."""
from core import SandboxTask
task = SandboxTask(
script="test.sh",
environment=[{"A": "1"}, {"B": "2"}, {"C": "3"}],
)
assert len(task.environment) == 3
assert task.environment[0] == {"A": "1"}
def test_global_vars_across_multiple_tasks(self, tmp_yaml):
"""Global vars resolve in both task_0 and task_1."""
from core import GlobalVariables, SandboxPipeline, SandboxTask0, SandboxTask1
t0 = SandboxTask0(
script="quantize.sh",
args=["--model", "<<global_vars.hf_model>>"],
environment=[{"HF_MODEL": "<<global_vars.hf_model>>"}],
)
t1 = SandboxTask1(
script="eval.sh",
environment=[{"HF_MODEL": "<<global_vars.hf_model>>"}],
)
pipeline = SandboxPipeline(
task_0=t0,
task_1=t1,
global_vars=GlobalVariables(hf_model="/hf-local/Qwen/Qwen3-8B"),
)
assert pipeline.tasks[0].args == ["--model", "/hf-local/Qwen/Qwen3-8B"]
assert pipeline.tasks[0].environment == [{"HF_MODEL": "/hf-local/Qwen/Qwen3-8B"}]
assert pipeline.tasks[1].environment == [{"HF_MODEL": "/hf-local/Qwen/Qwen3-8B"}]
class TestTestYamlFormat:
"""Tests for the test YAML format used by run_test_yaml.sh."""
def test_target_with_overrides(self, tmp_yaml):
"""Test YAML entries have _target_ and override fields."""
content = """
- _target_: path/to/config.yaml
pipeline:
allow_to_fail: true
skip: false
note: "known issue"
- _target_: path/to/other.yaml
pipeline:
allow_to_fail: false
"""
path = tmp_yaml(content)
with open(path) as f:
data = yaml.safe_load(f)
assert isinstance(data, list)
assert len(data) == 2
assert data[0]["_target_"] == "path/to/config.yaml"
assert data[0]["pipeline"]["allow_to_fail"] is True
assert data[0]["pipeline"]["note"] == "known issue"
assert data[1]["_target_"] == "path/to/other.yaml"
assert data[1]["pipeline"]["allow_to_fail"] is False
def test_flatten_overrides(self):
"""Nested overrides flatten to dot-notation for CLI args."""
entry = {
"pipeline": {
"allow_to_fail": True,
"skip": False,
}
}
# Simulate the flatten logic from run_test_yaml.sh
overrides = []
def flatten(d, prefix=""):
for k, v in d.items():
key = f"{prefix}{k}" if prefix else k
if isinstance(v, dict):
flatten(v, f"{key}.")
else:
overrides.append(f"{key}={v}")
flatten(entry)
assert "pipeline.allow_to_fail=True" in overrides
assert "pipeline.skip=False" in overrides