mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
``` # Install cd Model-Optimizer/launcher curl -LsSf https://astral.sh/uv/install.sh | sh git submodule update --init --recursive # Run locally with Docker (single GPU) uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml hf_local=/mnt/hf-local --yes # Run on Slurm cluster (no need to export the follow SLURM_XXX envs if used in sandbox) export SLURM_HOST=login-node.example.com export SLURM_ACCOUNT=my_account export SLURM_HF_LOCAL=/shared/hf-local export SLURM_JOB_DIR=/shared/experiments uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --yes # Preview config without running uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --dryrun --yes -v # Override parameters uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml \ pipeline.task_0.slurm_config.nodes=2 --yes # Dump resolved config for reproducibility (single YAML for reproducibility, great for QA, Eng, and agent to triage) uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --to-yaml resolved.yaml # Run tests uv pip install -e . pytest uv run pytest -v ``` ## Summary Add `launcher/` module for submitting quantization, training, and evaluation jobs to Slurm clusters or running them locally with Docker via `nemo-run`. `nemo-run` is used in all `NVIDIA-NeMo/*` projects. It supports modern YAML factory (superset of the `OmegaConf` and `Hydra`) and it support multiple executor backends (here we use docker and slurm mainly). A sample YAML config `launcher/Qwen/Qwen3-8B/megatron_lm_ptq.yaml`: ``` job_name: Qwen3-8B_NVFP4_DEFAULT_CFG pipeline: # hf_local: path prefix for model weights and datasets. # # This should be a self-managed directory that mirrors the HuggingFace Hub # hierarchy (e.g., /hf-local/Qwen/Qwen3-8B/, /hf-local/cais/mmlu/). Using # a dedicated folder is preferred over the HuggingFace cache (~/.cache/huggingface) # to avoid cache corruption issues with concurrent jobs. # # Override on CLI: # pipeline.global_vars.hf_local=/mnt/my-models/ # use a different path # pipeline.global_vars.hf_local="" # download from HuggingFace Hub global_vars: hf_local: /hf-local/ task_0: script: common/megatron-lm/quantize/quantize.sh args: - --calib-dataset-path-or-name <<global_vars.hf_local>>abisee/cnn_dailymail - --calib-size 32 environment: - MLM_MODEL_CFG: Qwen/Qwen3-8B - QUANT_CFG: NVFP4_DEFAULT_CFG - HF_MODEL_CKPT: <<global_vars.hf_local>>Qwen/Qwen3-8B - MMLU_DATASET: <<global_vars.hf_local>>cais/mmlu - TP: 4 slurm_config: _factory_: "slurm_factory" nodes: 1 ntasks_per_node: 4 gpus_per_node: 4 ``` ### Key features - **`launch.py`** — public entrypoint accepting `--yaml` config format - **`core.py`** — shared logic (dataclasses, executor builders, run loop) also used by nmm-sandbox's `slurm.py` - **Factory system** — env-var-driven `slurm_factory` with `register_factory()` registry - **`<<global_vars.X>>`** interpolation for sharing values across pipeline tasks - **`hf_local`** global var for configurable model/dataset storage path - **Version reporting** — git commit/branch printed at job start for reproducibility - **`--to-yaml`** — dump resolved config for bug reports and reproducibility - **Model-Optimizer symlink** — `modules/Model-Optimizer -> ../..` (auto-created, avoids recursive submodule) ### Files | Path | Description | |------|-------------| | `launcher/launch.py` | Public entrypoint | | `launcher/core.py` | Shared dataclasses, executors, run loop | | `launcher/slurm_config.py` | SlurmConfig + env-var factory | | `launcher/common/` | Shell scripts (quantize, query, eagle3, specdec_bench) | | `launcher/Qwen/Qwen3-8B/` | Example configs (PTQ, EAGLE3 pipeline) | | `launcher/tests/` | 64 unit tests | | `launcher/README.md` | User guide | | `launcher/ADVANCED.md` | Architecture, mount mechanism, Claude Code workflows | | `launcher/CLAUDE.md` | Claude Code project instructions | | `.github/workflows/unit_tests.yml` | CI job for launcher tests | ### Verified - Same YAML produces identical MMLU results via both `slurm.py` and `launch.py`: - Local Docker (TP=1): 0.719 (128/178) - OCI-HSG Slurm (TP=4): 0.730 (130/178) ## Test plan - [x] 64 unit tests (core, factory, YAML, Docker executor, Slurm executor, Docker launch) - [x] CI workflow added to `.github/workflows/unit_tests.yml` - [x] Local Docker end-to-end with `python:3.12-slim` - [x] Qwen3-8B PTQ on OCI-HSG via both launchers - [ ] Reviewer runs: `cd launcher && uv pip install -e . pytest && uv run pytest -v` ### Before your PR is "*Ready for review*" Make sure you read and follow [Contributor guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md) and your commits are signed (`git commit -s -S`). Make sure you read and follow the [Security Best Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors) (e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(..., weights_only=False)`, `pickle`, etc.). - Is this change backward compatible?: ✅ / ❌ / N/A <!--- If ❌, explain why. --> - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: ✅ / ❌ / N/A <!--- Mandatory --> - Did you write any new necessary tests?: ✅ / ❌ / N/A <!--- Mandatory for new features or examples. --> - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: ✅ / ❌ / N/A <!--- Only for new features, API changes, critical bug fixes or backward incompatible changes. --> ### Additional Information <!-- E.g. related issue. --> <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit ## Release Notes * **New Features** * Introduced ModelOpt Launcher for submitting quantization, training, and evaluation jobs to Slurm clusters or running locally via Docker. * Added YAML-based job configuration with multi-task pipeline support and global variable interpolation. * Included example workflows for Qwen3-8B quantization and EAGLE3 speculative decoding. * Provided configurable Slurm and execution environment defaults. * **Documentation** * Added comprehensive README with quick start, environment setup, and configuration guidance. * Added advanced guide detailing launcher architecture and integration patterns. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Chenhan Yu <chenhany@nvidia.com> Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
125 lines
3.9 KiB
Python
125 lines
3.9 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""Integration test for Docker container launch via run_jobs.
|
|
|
|
Requires Docker to be installed and running. Uses python:3.12-slim
|
|
(lightweight, no GPU needed) to run a trivial script.
|
|
|
|
Run with: pytest -s (stdin capture must be disabled for invoke/fabric)
|
|
"""
|
|
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
|
|
import pytest
|
|
|
|
docker_available = shutil.which("docker") is not None
|
|
|
|
|
|
@pytest.mark.skipif(not docker_available, reason="Docker not available")
|
|
class TestDockerLaunch:
|
|
"""End-to-end Docker launch test using subprocess to avoid pytest stdin capture issues."""
|
|
|
|
def test_echo_script_via_launch(self, tmp_path):
|
|
"""Launch a Docker container via launch.py subprocess that runs 'echo hello'."""
|
|
# Create a trivial script
|
|
script_dir = tmp_path / "scripts"
|
|
script_dir.mkdir()
|
|
script = script_dir / "hello.sh"
|
|
script.write_text("#!/bin/bash\necho 'HELLO_FROM_DOCKER'\n")
|
|
script.chmod(0o755)
|
|
|
|
# Create a YAML config
|
|
yaml_content = """
|
|
job_name: test_hello
|
|
pipeline:
|
|
task_0:
|
|
script: scripts/hello.sh
|
|
slurm_config:
|
|
_factory_: "slurm_factory"
|
|
container: python:3.12-slim
|
|
"""
|
|
yaml_path = tmp_path / "test.yaml"
|
|
yaml_path.write_text(yaml_content)
|
|
|
|
# Run launch.py as a subprocess (avoids pytest stdin capture issues)
|
|
launcher_dir = os.path.join(os.path.dirname(__file__), "..")
|
|
launcher_dir = os.path.abspath(launcher_dir)
|
|
|
|
result = subprocess.run(
|
|
[
|
|
"uv",
|
|
"run",
|
|
"launch.py",
|
|
"--yaml",
|
|
str(yaml_path),
|
|
f"hf_local={tmp_path}",
|
|
"--yes",
|
|
],
|
|
cwd=launcher_dir,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300,
|
|
)
|
|
|
|
# Check output
|
|
assert "Version Report" in result.stdout
|
|
assert "Launching" in result.stdout or "Entering Experiment" in result.stdout
|
|
|
|
def test_failing_script_via_launch(self, tmp_path):
|
|
"""Launch a Docker container that exits 1 — launch.py should not crash."""
|
|
script_dir = tmp_path / "scripts"
|
|
script_dir.mkdir()
|
|
script = script_dir / "fail.sh"
|
|
script.write_text("#!/bin/bash\necho 'FAILING'\nexit 1\n")
|
|
script.chmod(0o755)
|
|
|
|
yaml_content = """
|
|
job_name: test_fail
|
|
pipeline:
|
|
task_0:
|
|
script: scripts/fail.sh
|
|
slurm_config:
|
|
_factory_: "slurm_factory"
|
|
container: python:3.12-slim
|
|
"""
|
|
yaml_path = tmp_path / "fail_test.yaml"
|
|
yaml_path.write_text(yaml_content)
|
|
|
|
launcher_dir = os.path.join(os.path.dirname(__file__), "..")
|
|
launcher_dir = os.path.abspath(launcher_dir)
|
|
|
|
result = subprocess.run(
|
|
[
|
|
"uv",
|
|
"run",
|
|
"launch.py",
|
|
"--yaml",
|
|
str(yaml_path),
|
|
f"hf_local={tmp_path}",
|
|
"--yes",
|
|
],
|
|
cwd=launcher_dir,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300,
|
|
)
|
|
|
|
# launch.py should complete (exit 0) even if the job fails
|
|
# The job failure is reported in stdout
|
|
assert "Version Report" in result.stdout
|