Files
Model-Optimizer/tools/launcher/tests/test_docker_launch.py
T
Chenhan D. YuandClaude Opus 4.6 839fa3d658 add: ModelOpt Launcher for Slurm job submission (#1031)
```
# Install                                                                                                                                                                              
cd Model-Optimizer/launcher                                                                                                                                                            
curl -LsSf https://astral.sh/uv/install.sh | sh                                                                                                                                        
git submodule update --init --recursive                                                                                                                                                
                                                                                                                                                                                         
# Run locally with Docker (single GPU)                                                                                                                                                 
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml hf_local=/mnt/hf-local --yes                                                                                              
                                                                                                                                                                                         
# Run on Slurm cluster (no need to export the follow SLURM_XXX envs if used in sandbox)                                                                                                                                                              
export SLURM_HOST=login-node.example.com                                                                                                                                               
export SLURM_ACCOUNT=my_account                                                                                                                                                        
export SLURM_HF_LOCAL=/shared/hf-local                                                                                                                                                 
export SLURM_JOB_DIR=/shared/experiments                                                                                                                                               
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --yes                                                                                                                       
                                                                                                                                                                                         
# Preview config without running                                                                                                                                                       
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --dryrun --yes -v                                                                                                           
                                                                                                                                                                                         
# Override parameters                                                                                                                                                                  
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml \                                                                                                                           
      pipeline.task_0.slurm_config.nodes=2 --yes                                                                                                                                         
                                                                                                                                                                                         
# Dump resolved config for reproducibility (single YAML for reproducibility, great for QA, Eng, and agent to triage)
uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --to-yaml resolved.yaml                                                                                                     
                                                                                                                                                                                         
# Run tests
uv pip install -e . pytest                                                                                                                                                             
uv run pytest -v 
```
 ## Summary

Add `launcher/` module for submitting quantization, training, and
evaluation jobs to Slurm clusters or running them locally with Docker
via `nemo-run`. `nemo-run` is used in all `NVIDIA-NeMo/*` projects. It
supports modern YAML factory (superset of the `OmegaConf` and `Hydra`)
and it support multiple executor backends (here we use docker and slurm
mainly).

A sample YAML config `launcher/Qwen/Qwen3-8B/megatron_lm_ptq.yaml`:
```
job_name: Qwen3-8B_NVFP4_DEFAULT_CFG
pipeline:
  # hf_local: path prefix for model weights and datasets.
  #
  # This should be a self-managed directory that mirrors the HuggingFace Hub
  # hierarchy (e.g., /hf-local/Qwen/Qwen3-8B/, /hf-local/cais/mmlu/). Using
  # a dedicated folder is preferred over the HuggingFace cache (~/.cache/huggingface)
  # to avoid cache corruption issues with concurrent jobs.
  #
  # Override on CLI:
  #   pipeline.global_vars.hf_local=/mnt/my-models/   # use a different path
  #   pipeline.global_vars.hf_local=""                 # download from HuggingFace Hub
  global_vars:
    hf_local: /hf-local/

  task_0:
    script: common/megatron-lm/quantize/quantize.sh
    args:
      - --calib-dataset-path-or-name <<global_vars.hf_local>>abisee/cnn_dailymail
      - --calib-size 32
    environment:
      - MLM_MODEL_CFG: Qwen/Qwen3-8B
      - QUANT_CFG: NVFP4_DEFAULT_CFG
      - HF_MODEL_CKPT: <<global_vars.hf_local>>Qwen/Qwen3-8B
      - MMLU_DATASET: <<global_vars.hf_local>>cais/mmlu
      - TP: 4
    slurm_config:
      _factory_: "slurm_factory"
      nodes: 1
      ntasks_per_node: 4
      gpus_per_node: 4
```
  
### Key features
  - **`launch.py`** — public entrypoint accepting `--yaml` config format
- **`core.py`** — shared logic (dataclasses, executor builders, run
loop) also used by nmm-sandbox's `slurm.py`
- **Factory system** — env-var-driven `slurm_factory` with
`register_factory()` registry
- **`<<global_vars.X>>`** interpolation for sharing values across
pipeline tasks
- **`hf_local`** global var for configurable model/dataset storage path
- **Version reporting** — git commit/branch printed at job start for
reproducibility
- **`--to-yaml`** — dump resolved config for bug reports and
reproducibility
- **Model-Optimizer symlink** — `modules/Model-Optimizer -> ../..`
(auto-created, avoids recursive submodule)
  ### Files
| Path | Description |
  |------|-------------|
  | `launcher/launch.py` | Public entrypoint |
  | `launcher/core.py` | Shared dataclasses, executors, run loop |
| `launcher/slurm_config.py` | SlurmConfig + env-var factory |
| `launcher/common/` | Shell scripts (quantize, query, eagle3,
specdec_bench) |
| `launcher/Qwen/Qwen3-8B/` | Example configs (PTQ, EAGLE3 pipeline) |
| `launcher/tests/` | 64 unit tests |
  | `launcher/README.md` | User guide |
| `launcher/ADVANCED.md` | Architecture, mount mechanism, Claude Code
workflows |
| `launcher/CLAUDE.md` | Claude Code project instructions |
  | `.github/workflows/unit_tests.yml` | CI job for launcher tests |
### Verified

- Same YAML produces identical MMLU results via both `slurm.py` and
`launch.py`:
    - Local Docker (TP=1): 0.719 (128/178)
- OCI-HSG Slurm (TP=4): 0.730 (130/178)
  ## Test plan
- [x] 64 unit tests (core, factory, YAML, Docker executor, Slurm
executor, Docker launch)
  - [x] CI workflow added to `.github/workflows/unit_tests.yml`
- [x] Local Docker end-to-end with `python:3.12-slim`
- [x] Qwen3-8B PTQ on OCI-HSG via both launchers
- [ ] Reviewer runs: `cd launcher && uv pip install -e . pytest && uv
run pytest -v`

### Before your PR is "*Ready for review*"

Make sure you read and follow [Contributor
guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md)
and your commits are signed (`git commit -s -S`).

Make sure you read and follow the [Security Best
Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors)
(e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(...,
weights_only=False)`, `pickle`, etc.).

- Is this change backward compatible?: ✅ / ❌ / N/A <!--- If ❌, explain
why. -->
- If you copied code from any other sources or added a new PIP
dependency, did you follow guidance in `CONTRIBUTING.md`: ✅ / ❌ / N/A
<!--- Mandatory -->
- Did you write any new necessary tests?: ✅ / ❌ / N/A <!--- Mandatory
for new features or examples. -->
- Did you update
[Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?:
✅ / ❌ / N/A <!--- Only for new features, API changes, critical bug fixes
or backward incompatible changes. -->

### Additional Information
<!-- E.g. related issue. -->


<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

## Release Notes

* **New Features**
* Introduced ModelOpt Launcher for submitting quantization, training,
and evaluation jobs to Slurm clusters or running locally via Docker.
* Added YAML-based job configuration with multi-task pipeline support
and global variable interpolation.
* Included example workflows for Qwen3-8B quantization and EAGLE3
speculative decoding.
  * Provided configurable Slurm and execution environment defaults.

* **Documentation**
* Added comprehensive README with quick start, environment setup, and
configuration guidance.
* Added advanced guide detailing launcher architecture and integration
patterns.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->

---------

Signed-off-by: Chenhan Yu <chenhany@nvidia.com>
Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-18 15:32:15 -07:00

125 lines
3.9 KiB
Python

# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""Integration test for Docker container launch via run_jobs.
Requires Docker to be installed and running. Uses python:3.12-slim
(lightweight, no GPU needed) to run a trivial script.
Run with: pytest -s (stdin capture must be disabled for invoke/fabric)
"""
import os
import shutil
import subprocess
import pytest
docker_available = shutil.which("docker") is not None
@pytest.mark.skipif(not docker_available, reason="Docker not available")
class TestDockerLaunch:
"""End-to-end Docker launch test using subprocess to avoid pytest stdin capture issues."""
def test_echo_script_via_launch(self, tmp_path):
"""Launch a Docker container via launch.py subprocess that runs 'echo hello'."""
# Create a trivial script
script_dir = tmp_path / "scripts"
script_dir.mkdir()
script = script_dir / "hello.sh"
script.write_text("#!/bin/bash\necho 'HELLO_FROM_DOCKER'\n")
script.chmod(0o755)
# Create a YAML config
yaml_content = """
job_name: test_hello
pipeline:
task_0:
script: scripts/hello.sh
slurm_config:
_factory_: "slurm_factory"
container: python:3.12-slim
"""
yaml_path = tmp_path / "test.yaml"
yaml_path.write_text(yaml_content)
# Run launch.py as a subprocess (avoids pytest stdin capture issues)
launcher_dir = os.path.join(os.path.dirname(__file__), "..")
launcher_dir = os.path.abspath(launcher_dir)
result = subprocess.run(
[
"uv",
"run",
"launch.py",
"--yaml",
str(yaml_path),
f"hf_local={tmp_path}",
"--yes",
],
cwd=launcher_dir,
capture_output=True,
text=True,
timeout=300,
)
# Check output
assert "Version Report" in result.stdout
assert "Launching" in result.stdout or "Entering Experiment" in result.stdout
def test_failing_script_via_launch(self, tmp_path):
"""Launch a Docker container that exits 1 — launch.py should not crash."""
script_dir = tmp_path / "scripts"
script_dir.mkdir()
script = script_dir / "fail.sh"
script.write_text("#!/bin/bash\necho 'FAILING'\nexit 1\n")
script.chmod(0o755)
yaml_content = """
job_name: test_fail
pipeline:
task_0:
script: scripts/fail.sh
slurm_config:
_factory_: "slurm_factory"
container: python:3.12-slim
"""
yaml_path = tmp_path / "fail_test.yaml"
yaml_path.write_text(yaml_content)
launcher_dir = os.path.join(os.path.dirname(__file__), "..")
launcher_dir = os.path.abspath(launcher_dir)
result = subprocess.run(
[
"uv",
"run",
"launch.py",
"--yaml",
str(yaml_path),
f"hf_local={tmp_path}",
"--yes",
],
cwd=launcher_dir,
capture_output=True,
text=True,
timeout=300,
)
# launch.py should complete (exit 0) even if the job fails
# The job failure is reported in stdout
assert "Version Report" in result.stdout