mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? **Type of change:** New feature (AutoQuantize recipes). The `--auto_quantize_*` CLI flags are **deprecated but still work** (kept as a thin backward-compat shim) — **not** removed. Makes **AutoQuantize recipe-driven**: `mtq.auto_quantize` is configured by a declarative YAML recipe (`--recipe`). The old `--auto_quantize_*` flags are converted into an `AutoQuantizeConfig` **on the fly** and run the exact same recipe path (emitting a `DeprecationWarning`), so old commands keep working. The recipe path is verified **byte-identical** to the CLI. - **Cost model** (`quantization/config.py`, `algorithms.py`): new `effective_bits` field on `QuantizeConfig` (recipe-level override) and `QuantizerAttributeConfig` (per-format default). `estimate_quant_compression` resolves recipe-level → per-entry → `num_bits` heuristic. `configs/numerics/nvfp4.yaml` ships `effective_bits: 4.5` (block-scale-accurate) as the single source of truth. - **Recipe schema** (`recipe/config.py`, `recipe/loader.py`): `RecipeType.AUTO_QUANTIZE` + `AutoQuantizeConfig` / `AutoQuantizeConstraints` / `AutoQuantizeCost`. Fields: `constraints` (`effective_bits`, `cost_model`, `cost.active_moe_expert_ratio`), `candidate_formats`, `auto_quantize_method` (`gradient`/`kl_div`), `score_size`, `disabled_layers`, `cost_excluded_layers` (e.g. VL vision towers), `kv_cache`. - **Dispatch** (`examples/hf_ptq/hf_ptq.py`): recipe → mtq inputs via `_mtq_inputs_from_auto_quantize_config`; `_match_candidate_to_preset` resolves candidates to shipped presets and **guards export-compatibility** (rejects export-unsafe presets before the search). - **Deprecated CLI shim:** `_auto_quantize_config_from_cli` builds an `AutoQuantizeConfig` from the old flags and appends the shared base `disabled_layers` / `cost_excluded_layers` (loaded once as module constants in `recipe/config.py`, mirroring `_default_disabled_quantizer_cfg`). No model introspection, no new user flags. - **Shipped recipes:** `general/auto_quantize/` (`nvfp4_fp8_at_5p4bits`, `nvfp4_fp8_kl_div_at_5p4bits`, `nvfp4_mse_fp8_at_6p0bits`, `w4a8_awq_beta_fp8_at_6p0bits`, `w4a16_nvfp4_fp8_at_6p0bits-active_moe`) and model-specific `huggingface/qwen3_6_moe/auto_quantize/...`. Shared `configs/auto_quantize/units/base_disabled_layers` + `base_cost_excluded_layers` spliced via `$import`. **Migration (deprecated flag → recipe field):** `--auto_quantize_bits` → `constraints.effective_bits` · `--auto_quantize_method` → `auto_quantize_method` · `--auto_quantize_score_size` → `score_size` · `--auto_quantize_cost_model` → `constraints.cost_model` · `--auto_quantize_active_moe_expert_ratio` → `constraints.cost.active_moe_expert_ratio` · `--qformat fp8,nvfp4` → `candidate_formats`. `--auto_quantize_checkpoint` unchanged. ### Usage ```sh # Recipe (preferred) python examples/hf_ptq/hf_ptq.py --pyt_ckpt_path <model> --recipe general/auto_quantize/nvfp4_fp8_at_5p4bits --export_path <out> # Deprecated CLI (converted to a recipe on the fly, still works) python examples/hf_ptq/hf_ptq.py --pyt_ckpt_path <model> --qformat nvfp4,fp8 --auto_quantize_bits 5.4 --export_path <out> ``` ### Testing - **GPU-free unit tests:** recipe loader; recipe→`mtq.auto_quantize` mapping incl. `cost_excluded_layers`; export-compat guard (reject/warn/no-bypass); deprecated-CLI→`AutoQuantizeConfig` conversion; `effective_bits` resolver + validators. - **Byte-identical export smoke:** recipe path on Qwen3.6-35B-A3B (`fp8 + w4a16_nvfp4 @ 6.0`, `active_moe`) → identical `hf_quant_config.json` across CLI/recipe; also confirmed the deprecated **CLI shim ≡ recipe** on the same VL MoE. ### Before your PR is "*Ready for review*" - Is this change backward compatible?: ✅ **Yes** — `--auto_quantize_*` flags are deprecated but still work (converted to a recipe on the fly + `DeprecationWarning`). Plain PTQ CLI unaffected. - New PIP dependency / copied code: N/A - New tests?: ✅ - Updated Changelog?: ✅ (Deprecations) - Claude approval?: pending `/claude review` 🤖 Generated with [Claude Code](https://claude.com/claude-code) --------- Signed-off-by: Juhi Mittal <juhim@nvidia.com> Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
103 lines
3.3 KiB
Python
103 lines
3.3 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2023-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
import importlib.metadata as metadata
|
|
import subprocess
|
|
from dataclasses import asdict, dataclass
|
|
|
|
import pytest
|
|
import torch
|
|
from _test_utils.examples.run_command import run_hf_ptq_command
|
|
|
|
|
|
@dataclass
|
|
class PTQCommand:
|
|
quant: str | None = None
|
|
recipe: str | None = None
|
|
tasks: str = "quant"
|
|
calib: int = 16
|
|
sparsity: str | None = None
|
|
kv_cache_quant: str | None = None
|
|
trust_remote_code: bool = False
|
|
calib_dataset: str = "cnn_dailymail"
|
|
calib_batch_size: int | None = None
|
|
tp: int | None = None
|
|
pp: int | None = None
|
|
min_sm: int | None = None
|
|
max_sm: int | None = None
|
|
min_gpu: int | None = None
|
|
batch: int | None = None
|
|
|
|
def __post_init__(self):
|
|
if (self.quant is None) == (self.recipe is None):
|
|
raise ValueError("Exactly one of `quant` or `recipe` must be set.")
|
|
|
|
def run(self, model_path: str):
|
|
if self.min_sm and torch.cuda.get_device_capability() < (
|
|
self.min_sm // 10,
|
|
self.min_sm % 10,
|
|
):
|
|
pytest.skip(reason=f"Requires sm{self.min_sm} or higher")
|
|
|
|
if self.max_sm and torch.cuda.get_device_capability() > (
|
|
self.max_sm // 10,
|
|
self.max_sm % 10,
|
|
):
|
|
pytest.skip(reason=f"Requires sm{self.max_sm} or lower")
|
|
|
|
if self.min_gpu and torch.cuda.device_count() < self.min_gpu:
|
|
pytest.skip(reason=f"Requires at least {self.min_gpu} GPUs")
|
|
|
|
param_dict = asdict(self)
|
|
param_dict.pop("min_sm", None)
|
|
param_dict.pop("max_sm", None)
|
|
param_dict.pop("min_gpu", None)
|
|
|
|
quant = param_dict.pop("quant")
|
|
run_hf_ptq_command(model=model_path, quant=quant, **param_dict)
|
|
|
|
def param_str(self):
|
|
param_dict = asdict(self)
|
|
param_dict.pop("trust_remote_code", False)
|
|
return "_".join(str(value) for value in param_dict.values() if value is not None).replace(
|
|
",", "_"
|
|
)
|
|
|
|
|
|
class WithRequirements:
|
|
requirements = []
|
|
|
|
@pytest.fixture(scope="class", autouse=True)
|
|
def install(self):
|
|
save_deps = []
|
|
for mod, ver in self.requirements:
|
|
try:
|
|
save_ver = metadata.version(mod)
|
|
except metadata.PackageNotFoundError:
|
|
save_ver = None
|
|
|
|
save_deps.append((mod, save_ver))
|
|
|
|
spec = f"{mod}=={ver}" if ver else mod
|
|
subprocess.run(["pip", "install", spec], check=True)
|
|
|
|
yield
|
|
|
|
for mod, ver in save_deps:
|
|
if ver:
|
|
subprocess.run(["pip", "install", f"{mod}=={ver}"], check=True)
|
|
else:
|
|
subprocess.run(["pip", "uninstall", "--yes", mod], check=True)
|