From 87ea8babe1da439c9eeff29367c689d1129e22cc Mon Sep 17 00:00:00 2001 From: Chenjie Luo <108829653+cjluo-nv@users.noreply.github.com> Date: Thu, 2 Apr 2026 11:55:38 -0700 Subject: [PATCH] Add HuggingFace PTQ pipeline to launcher (#1100) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### What does this PR do? Type of change: New feature Adds a HuggingFace PTQ pipeline to the launcher, replacing the old `hf_ptq.sh`/`hf_ptq_local.yaml` approach with a cleaner wrapper around `huggingface_example.sh`. **Key changes:** - **New `common/hf/ptq.sh`** — wrapper script that downloads the model via `huggingface-cli` if needed, then delegates to `examples/llm_ptq/scripts/huggingface_example.sh` - **New `examples/Qwen/Qwen3-8B/hf_ptq.yaml`** — example config for Qwen3-8B nvfp4 quantization, supports both Slurm and local Docker - **Removed `common/hf_ptq/hf_ptq.sh`** and **`examples/Qwen/Qwen3-8B/hf_ptq_local.yaml`** — replaced by the new unified pipeline - **Configurable Slurm time limit** — `SlurmConfig.time` field replaces the hardcoded `"04:00:00"` in `build_slurm_executor` - **Configurable Slurm partition** — `slurm_factory` now reads `SLURM_PARTITION` env var (default: `batch`) - **`--clean` flag** — new `launch.py` option to `git clean -xdf` the examples directory before job submission - **Package `modelopt_recipes/`** — added to the nemo_run packager include list ### Testing - Tested HF PTQ pipeline on Slurm with Qwen3-8B ### Before your PR is "*Ready for review*" Make sure you read and follow [Contributor guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md) and your commits are signed (`git commit -s -S`). Make sure you read and follow the [Security Best Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors) (e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(..., weights_only=False)`, `pickle`, etc.). - Is this change backward compatible?: ✅ - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: N/A - Did you write any new necessary tests?: ❌ - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: N/A ## Summary by CodeRabbit * **New Features** * Added Hugging Face PTQ workflow configuration for Qwen model quantization. * Added `clean` parameter to launcher for clearing directories before job execution. * **Improvements** * Made Slurm execution time configurable per job instead of hardcoded values. * Slurm partition configuration now respects environment variables. * **Deprecated** * Removed legacy PTQ launcher scripts, replaced with unified wrapper for improved maintainability. --------- Signed-off-by: Chenjie Luo --- tools/launcher/common/hf/ptq.sh | 62 +++++++++++++++++++ tools/launcher/common/hf_ptq/hf_ptq.sh | 39 ------------ tools/launcher/core.py | 2 +- .../examples/Qwen/Qwen3-8B/hf_ptq.yaml | 51 +++++++++++++++ .../examples/Qwen/Qwen3-8B/hf_ptq_local.yaml | 27 -------- tools/launcher/launch.py | 10 ++- tools/launcher/slurm_config.py | 5 +- tools/launcher/tests/test_slurm_executor.py | 1 + 8 files changed, 128 insertions(+), 69 deletions(-) create mode 100755 tools/launcher/common/hf/ptq.sh delete mode 100644 tools/launcher/common/hf_ptq/hf_ptq.sh create mode 100644 tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq.yaml delete mode 100644 tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq_local.yaml diff --git a/tools/launcher/common/hf/ptq.sh b/tools/launcher/common/hf/ptq.sh new file mode 100755 index 000000000..b3bc80c30 --- /dev/null +++ b/tools/launcher/common/hf/ptq.sh @@ -0,0 +1,62 @@ +#!/bin/bash +# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# HuggingFace PTQ wrapper: downloads the model if needed, then runs huggingface_example.sh. +# +# Usage: +# ptq.sh --repo --local-dir -- [huggingface_example.sh args...] +# +# Everything before "--" is handled by this wrapper (download logic). +# Everything after "--" is passed directly to huggingface_example.sh. +# The --model arg is automatically set to for huggingface_example.sh. + +set -e + +REPO="" +LOCAL_DIR="" +PTQ_ARGS=() + +# Parse wrapper args up to "--", collect the rest for huggingface_example.sh +while [[ $# -gt 0 ]]; do + case "$1" in + --repo) REPO="$2"; shift 2 ;; + --local-dir) LOCAL_DIR="$2"; shift 2 ;; + --) shift; PTQ_ARGS=("$@"); break ;; + *) echo "Unknown argument: $1 (use -- to separate PTQ args)" >&2; exit 1 ;; + esac +done + +if [ -z "$REPO" ] || [ -z "$LOCAL_DIR" ]; then + echo "Usage: ptq.sh --repo --local-dir -- [huggingface_example.sh args...]" >&2 + exit 1 +fi + +# --- Step 1: Download model if not already present --- +if [ -f "$LOCAL_DIR/config.json" ]; then + echo "Model already exists at $LOCAL_DIR, skipping download." +else + echo "Downloading $REPO to $LOCAL_DIR ..." + pip install -q huggingface_hub 2>/dev/null || true + huggingface-cli download "$REPO" --local-dir "$LOCAL_DIR" + echo "Download complete: $LOCAL_DIR" +fi + +# --- Step 2: Run huggingface_example.sh --- +script_dir="$(dirname "$(readlink -f "$0")")" +HF_EXAMPLE="${script_dir}/../../modules/Model-Optimizer/examples/llm_ptq/scripts/huggingface_example.sh" + +echo "Running huggingface_example.sh --model $LOCAL_DIR --trust_remote_code ${PTQ_ARGS[*]}" +exec bash "$HF_EXAMPLE" --model "$LOCAL_DIR" --trust_remote_code "${PTQ_ARGS[@]}" diff --git a/tools/launcher/common/hf_ptq/hf_ptq.sh b/tools/launcher/common/hf_ptq/hf_ptq.sh deleted file mode 100644 index 8d53ebb04..000000000 --- a/tools/launcher/common/hf_ptq/hf_ptq.sh +++ /dev/null @@ -1,39 +0,0 @@ -#!/bin/bash - -# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -SCRIPT_DIR="$(dirname "$(readlink -f "$0")")" -source "${SCRIPT_DIR}/../service_utils.sh" - -trap 'error_handler $0 $LINENO' ERR # ERROR HANDLER -trap 'exit_handler' EXIT -################################################################################################### - -HF_PTQ_DIR=modules/Model-Optimizer/examples/llm_ptq - -HF_MODEL=${HF_MODEL:-"Qwen/Qwen3-8B"} -QFORMAT=${QFORMAT:-"fp8"} -CALIB_SIZE=${CALIB_SIZE:-"512"} -EXPORT_PATH=${EXPORT_PATH:-"/scratchspace/exported_model"} - -PYTHONPATH="${HF_PTQ_DIR}:${PYTHONPATH}" python ${HF_PTQ_DIR}/hf_ptq.py \ - --pyt_ckpt_path ${HF_MODEL} \ - --qformat ${QFORMAT} \ - --calib_size ${CALIB_SIZE} \ - --export_path ${EXPORT_PATH} \ - "$@" - -report_result "PASS: hf_ptq ${HF_MODEL} ${QFORMAT}" diff --git a/tools/launcher/core.py b/tools/launcher/core.py index 40e6c9441..7004f7e66 100644 --- a/tools/launcher/core.py +++ b/tools/launcher/core.py @@ -268,7 +268,7 @@ def build_slurm_executor( container_image=slurm_config.container, container_mounts=container_mounts, array=slurm_config.array, - time="04:00:00", + time=slurm_config.time, mem="0", retries=0, packager=packager, diff --git a/tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq.yaml b/tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq.yaml new file mode 100644 index 000000000..f8c1316f1 --- /dev/null +++ b/tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq.yaml @@ -0,0 +1,51 @@ +# HuggingFace PTQ via huggingface_example.sh +# +# Quantizes a HuggingFace model using examples/llm_ptq/scripts/huggingface_example.sh. +# Default: Qwen/Qwen3.5-9B with nvfp4_mlp_only on 8xH200. +# +# Usage (Slurm): +# export SLURM_HOST= +# export SLURM_ACCOUNT= +# export SLURM_PARTITION= # default: batch +# export SLURM_JOB_DIR=/home/scratch./experiments +# export SLURM_HF_LOCAL=/home/scratch./hf-local +# export HF_TOKEN= # for gated models; auto-injected into all tasks +# cd tools/launcher +# uv run launch.py --yaml examples/llm_ptq/hf_ptq.yaml --yes +# +# Usage (local Docker): +# cd tools/launcher +# uv run launch.py --yaml examples/llm_ptq/hf_ptq.yaml hf_local=/mnt/hf-local --yes +# +# Override model/quant via CLI: +# uv run launch.py --yaml examples/llm_ptq/hf_ptq.yaml \ +# pipeline.global_vars.hf_model=Qwen/Qwen3-8B \ +# pipeline.task_0.args='[--model,<>Qwen/Qwen3-8B,--quant,nvfp4]' \ +# --yes + +job_name: hf_ptq_nvfp4 +pipeline: + skip: false + allow_to_fail: false + note: "HF PTQ with nvfp4" + + global_vars: + hf_local: /hf-local/ + hf_model: Qwen/Qwen3-8B + + # Downloads model if needed, then runs huggingface_example.sh + task_0: + script: common/hf/ptq.sh + args: + - --repo <> + - --local-dir <><> + - -- + - --quant nvfp4 + - --tasks quant + slurm_config: + _factory_: "slurm_factory" + nodes: 1 + ntasks_per_node: 1 + gpus_per_node: 1 + time: "04:00:00" + container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc7 diff --git a/tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq_local.yaml b/tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq_local.yaml deleted file mode 100644 index 17a806bda..000000000 --- a/tools/launcher/examples/Qwen/Qwen3-8B/hf_ptq_local.yaml +++ /dev/null @@ -1,27 +0,0 @@ -# Local single-GPU HF PTQ for Qwen3-8B using hf_ptq.py from Model-Optimizer. -# -# Runs hf_ptq.py directly (Hugging Face path, no Megatron-LM conversion). -# -# Usage: -# uv run launch.py --yaml examples/Qwen/Qwen3-8B/hf_ptq_local.yaml hf_local=/mnt/hf-local --yes - -job_name: Qwen3-8B_fp8_hf_ptq_local -pipeline: - skip: false - allow_to_fail: false - note: - - task_0: - script: common/hf_ptq/hf_ptq.sh - args: - - --dataset cnn_dailymail - environment: - - HF_MODEL: /hf-local/Qwen/Qwen3-8B - - QFORMAT: fp8 - - CALIB_SIZE: "512" - - EXPORT_PATH: /scratchspace/exported_model - slurm_config: - _factory_: "slurm_factory" - nodes: 1 - ntasks_per_node: 1 - gpus_per_node: 1 diff --git a/tools/launcher/launch.py b/tools/launcher/launch.py index 6572a447f..3447d1faa 100644 --- a/tools/launcher/launch.py +++ b/tools/launcher/launch.py @@ -30,6 +30,7 @@ Environment variables: import getpass import os +import subprocess # nosec B404 import warnings import nemo_run as run @@ -61,10 +62,11 @@ packager = run.PatternPackager( "modules/Megatron-LM/examples/*", "modules/Megatron-LM/*.py", "modules/Model-Optimizer/modelopt/*", + "modules/Model-Optimizer/modelopt_recipes/*", "modules/Model-Optimizer/examples/*", "common/*", ], - relative_path=[LAUNCHER_DIR] * 6, + relative_path=[LAUNCHER_DIR] * 7, ) MODELOPT_SRC_PATH = os.path.join(LAUNCHER_DIR, "modules/Model-Optimizer/modelopt") @@ -84,8 +86,14 @@ def launch( user: str = getpass.getuser(), identity: str = None, # noqa: RUF013 detach: bool = False, + clean: bool = False, ) -> None: """Launch ModelOpt jobs on Slurm or locally with Docker.""" + if clean: + examples_dir = os.path.join(_mo_symlink, "examples") + print(f"Cleaning {examples_dir} with git clean -xdf ...") + subprocess.run(["git", "clean", "-xdf", "."], cwd=examples_dir, check=True) # nosec B603 B607 + if "NEMORUN_HOME" not in os.environ: warnings.warn("NEMORUN_HOME is not set. Defaulting to current working directory.") run.config.set_nemorun_home(os.environ.get("NEMORUN_HOME", os.getcwd())) diff --git a/tools/launcher/slurm_config.py b/tools/launcher/slurm_config.py index 0bf52d194..d2a8cd48d 100644 --- a/tools/launcher/slurm_config.py +++ b/tools/launcher/slurm_config.py @@ -41,6 +41,7 @@ class SlurmConfig: nodes: int = 1 ntasks_per_node: int = 1 gpus_per_node: int = 1 + time: str = "04:00:00" local: bool = False @@ -49,7 +50,7 @@ class SlurmConfig: def slurm_factory( host: str = os.environ.get("SLURM_HOST", ""), account: str = os.environ.get("SLURM_ACCOUNT", ""), - partition: str = "batch", + partition: str = os.environ.get("SLURM_PARTITION", "batch"), nodes: int = 1, ntasks_per_node: int = 1, gpus_per_node: int = 1, @@ -60,6 +61,7 @@ def slurm_factory( ], srun_args: list[str] = ["--no-container-mount-home"], array: str = None, # noqa: RUF013 + time: str = "04:00:00", ) -> SlurmConfig: """Generic Slurm factory — configure via environment variables or CLI overrides.""" return SlurmConfig( @@ -74,4 +76,5 @@ def slurm_factory( container_mounts=container_mounts, srun_args=srun_args, array=array, + time=time, ) diff --git a/tools/launcher/tests/test_slurm_executor.py b/tools/launcher/tests/test_slurm_executor.py index d7ac7827f..5f2d1b8da 100644 --- a/tools/launcher/tests/test_slurm_executor.py +++ b/tools/launcher/tests/test_slurm_executor.py @@ -168,6 +168,7 @@ class TestBuildSlurmExecutor: ntasks_per_node=8, gpus_per_node=8, array="0-3", + time="04:00:00", ) packager = MagicMock()