mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
## Summary - add IQ format metadata and packed-weight export - support Hugging Face and TP=1 Megatron export paths - reject fused-MoE IQ export until a deployment loader owns its packed layout - document the shaped `uint8` weight contract and the fused-expert boundary - add Hugging Face, Megatron, metadata, and fused-expert export tests ## PR split This work is split into four focused PRs. Each PR targets `main` and owns a disjoint file set: 1. **Kernel** — [#2448: Add CUDA kernels for IQ packing](https://github.com/NVIDIA/Model-Optimizer/pull/2448) 2. **Quantization** — [#2446: Add IQ quantization codecs and backend](https://github.com/NVIDIA/Model-Optimizer/pull/2446) 3. **Export** — [#2447: Export IQ checkpoints from HF and Megatron](https://github.com/NVIDIA/Model-Optimizer/pull/2447) 4. **Recipes** — [#2449: Add IQ post-training quantization recipes](https://github.com/NVIDIA/Model-Optimizer/pull/2449) The required merge order is #2448, #2446, #2447, then #2449. ## Scope This PR owns only export code, deployment documentation, and export tests. It targets `main` and should merge after #2448 and #2446. It does not contain kernel, codec/backend, or recipe files. ## Deployment consumer boundary Dense weights and individually named expert weights use the documented shaped `uint8` contract. Megatron fused-MoE IQ export is intentionally rejected with `NotImplementedError`: its payload would have shape `[num_experts, out_features, in_features // 256, payload_bytes]`, and no deployment loader in this stack currently owns that layout. Support should be enabled only with a loader integration test. ## Test coverage - [Hugging Face packed-weight export](https://github.com/NVIDIA/Model-Optimizer/blob/11cd58d907465933f5a552bc1a8065f84c9ba3b1/tests/unit/torch/export/test_export_weight.py) - [quantization metadata](https://github.com/NVIDIA/Model-Optimizer/blob/11cd58d907465933f5a552bc1a8065f84c9ba3b1/tests/unit/torch/export/test_get_quantization.py) - [Megatron unified export and fused-MoE rejection](https://github.com/NVIDIA/Model-Optimizer/blob/11cd58d907465933f5a552bc1a8065f84c9ba3b1/tests/gpu_megatron/torch/export/test_unified_export_megatron.py) ## Validation - all pre-commit hooks pass for the changed files - 89 focused Hugging Face export, metadata, and fused-expert tests pass locally - direct checks cover both fused-MoE export entry points for IQ1_S and IQ2_XS - Megatron GPU execution remains delegated to GPU CI - restricted-term scan passes <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added support for IQ1_S and IQ2_XS GGML quantization formats in unified Hugging Face and Megatron exports. * Added quantization metadata, tensor-shape recovery, packing details, and IQ2_XS size documentation. * Added validation for required block sizes and tensor parallelism settings. * **Limitations** * Fused-MoE and GPT-OSS IQ expert packing are not supported. * IQ exports require standard `weight` attributes in Hugging Face models. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Hung-Yueh Chiang <hungyuehc@nvidia.com> Signed-off-by: Chenjie Luo <chenjiel@nvidia.com> Co-authored-by: Chenjie Luo <chenjiel@nvidia.com> Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
329 lines
13 KiB
Python
329 lines
13 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""Convert modelopt quantization export config to align with llm-compressor config format."""
|
|
|
|
import warnings
|
|
from collections import defaultdict
|
|
from typing import Any
|
|
|
|
from modelopt.torch.quantization.ggml import (
|
|
IQ1_S_BLOCK_BYTES,
|
|
IQ1_S_BLOCK_SIZE,
|
|
IQ1_S_EFFECTIVE_BITS,
|
|
IQ2_XS_BLOCK_BYTES,
|
|
IQ2_XS_BLOCK_SIZE,
|
|
IQ2_XS_EFFECTIVE_BITS,
|
|
)
|
|
|
|
|
|
def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) -> dict[str, Any]:
|
|
"""Map a per-layer quant_algo string to compressed-tensors config group details.
|
|
|
|
Args:
|
|
quant_algo: The quantization algorithm name (e.g. "FP8", "NVFP4").
|
|
group_size: Optional group size for block-wise quantization algorithms.
|
|
|
|
Returns:
|
|
Dictionary with ``input_activations`` and ``weights`` entries suitable for
|
|
a compressed-tensors ``config_groups`` entry, or ModelOpt-owned metadata for
|
|
self-contained IQ payloads.
|
|
"""
|
|
if quant_algo == "FP8":
|
|
return {
|
|
"input_activations": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
"weights": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
}
|
|
elif quant_algo == "FP8_PER_CHANNEL_PER_TOKEN":
|
|
return {
|
|
"input_activations": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
"weights": {"dynamic": False, "num_bits": 8, "type": "float", "strategy": "channel"},
|
|
}
|
|
elif quant_algo == "NVFP4":
|
|
gs = group_size or 16
|
|
return {
|
|
"input_activations": {
|
|
"dynamic": False,
|
|
"num_bits": 4,
|
|
"type": "float",
|
|
"group_size": gs,
|
|
},
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": gs},
|
|
}
|
|
elif quant_algo == "W4A16_AWQ":
|
|
gs = group_size or 128
|
|
return {
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "int", "group_size": gs},
|
|
}
|
|
elif quant_algo == "W4A16_NVFP4":
|
|
gs = group_size or 16
|
|
return {
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": gs},
|
|
}
|
|
elif quant_algo == "NVFP4_SVD":
|
|
gs = group_size or 16
|
|
return {
|
|
"input_activations": {
|
|
"dynamic": False,
|
|
"num_bits": 4,
|
|
"type": "float",
|
|
"group_size": gs,
|
|
},
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": gs},
|
|
"has_zero_point": False,
|
|
"pre_quant_scale": True,
|
|
}
|
|
elif quant_algo in ("NVFP4_AWQ", "W4A8_AWQ"):
|
|
gs = group_size or 128
|
|
return {
|
|
"input_activations": {
|
|
"dynamic": False,
|
|
"num_bits": 8,
|
|
"type": "float",
|
|
"group_size": gs,
|
|
},
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": gs},
|
|
}
|
|
elif quant_algo == "W8A16":
|
|
return {
|
|
"weights": {"dynamic": False, "num_bits": 8, "type": "int"},
|
|
}
|
|
elif quant_algo == "W8A8_SQ_PER_CHANNEL":
|
|
return {
|
|
"input_activations": {"dynamic": False, "num_bits": 8, "type": "int"},
|
|
"weights": {
|
|
"dynamic": False,
|
|
"num_bits": 8,
|
|
"type": "int",
|
|
"strategy": "channel",
|
|
},
|
|
}
|
|
elif quant_algo in ("W4A8_NVFP4_FP8", "W4A8_MXFP4_FP8"):
|
|
gs = group_size or 16
|
|
return {
|
|
"input_activations": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": gs},
|
|
}
|
|
elif quant_algo == "MXFP8":
|
|
gs = group_size or 32
|
|
return {
|
|
"input_activations": {
|
|
"dynamic": False,
|
|
"num_bits": 8,
|
|
"type": "float",
|
|
"group_size": gs,
|
|
},
|
|
"weights": {"dynamic": False, "num_bits": 8, "type": "float", "group_size": gs},
|
|
}
|
|
elif quant_algo in ("IQ1_S", "IQ2_XS"):
|
|
if quant_algo == "IQ1_S":
|
|
block_size = IQ1_S_BLOCK_SIZE
|
|
payload_bytes = IQ1_S_BLOCK_BYTES
|
|
effective_bits = IQ1_S_EFFECTIVE_BITS
|
|
else:
|
|
block_size = IQ2_XS_BLOCK_SIZE
|
|
payload_bytes = IQ2_XS_BLOCK_BYTES
|
|
effective_bits = IQ2_XS_EFFECTIVE_BITS
|
|
if group_size not in (None, block_size):
|
|
raise ValueError(f"{quant_algo} requires group size {block_size}, got {group_size}")
|
|
# IQ payloads are self-contained blocks, not compressed-tensors integer groups.
|
|
# Keep their format marker outside a ``weights`` quantization scheme.
|
|
return {
|
|
"quant_algo": quant_algo,
|
|
"effective_bits": effective_bits,
|
|
"group_size": block_size,
|
|
"packing": "ggml",
|
|
"block_payload_bytes": payload_bytes,
|
|
}
|
|
else:
|
|
warnings.warn(
|
|
f"Unsupported quantization algorithm '{quant_algo}' in "
|
|
f"_quant_algo_to_group_config. The resulting config group will not contain "
|
|
f"'input_activations' or 'weights' keys and may not be compatible with "
|
|
f"compressed-tensors consumers. Please add explicit support for this algorithm."
|
|
)
|
|
return {"quant_algo": quant_algo}
|
|
|
|
|
|
def convert_hf_quant_config_format(input_config: dict[str, Any]) -> dict[str, Any]:
|
|
"""Converts modelopt quantization config dictionary to align with llm-compressor config format.
|
|
|
|
Args:
|
|
input_config: The original quantization config dictionary.
|
|
|
|
Note:
|
|
The "targets" field specifies which PyTorch module types to quantize. Compressed-tensors
|
|
works with any PyTorch module type and uses dynamic matching against module.__class__.__name__.
|
|
Typically this includes "Linear" modules, but can also include "Embedding" and other types.
|
|
|
|
See: https://github.com/neuralmagic/compressed-tensors/blob/fa6a48f1da6b47106912bcd25eba7171ba7cfec7/src/sparsetensors/quantization/quant_scheme.py#L29
|
|
Example usage: https://github.com/neuralmagic/compressed-tensors/blob/9938a6ec6e10498d39a3071dfd1c40e3939ee80b/tests/test_quantization/lifecycle/test_apply.py#L118
|
|
|
|
Example:
|
|
|
|
.. code-block:: python
|
|
|
|
{
|
|
"producer": {"name": "modelopt", "version": "0.19.0"},
|
|
"quantization": {
|
|
"quant_algo": "FP8",
|
|
"kv_cache_quant_algo": "FP8",
|
|
"exclude_modules": ["lm_head"],
|
|
},
|
|
}
|
|
|
|
Returns:
|
|
A new dictionary in the target format.
|
|
|
|
Example (for FP8 input):
|
|
|
|
.. code-block:: python
|
|
|
|
{
|
|
"config_groups": {
|
|
"group_0": {
|
|
"input_activations": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
"weights": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
}
|
|
},
|
|
"ignore": ["lm_head"],
|
|
"quant_algo": "FP8",
|
|
"kv_cache_scheme": "FP8",
|
|
"producer": {"name": "modelopt", "version": "0.29.0"},
|
|
}
|
|
"""
|
|
new_config: dict[str, Any] = {}
|
|
|
|
original_quantization_details = input_config.get("quantization", {})
|
|
quant_algo_value = original_quantization_details.get("quant_algo")
|
|
|
|
# This structure is derived based on the example for "FP8" and "NVFP4"
|
|
# TODO: Handle other quantization algorithms
|
|
if quant_algo_value == "FP8":
|
|
config_group_details: dict[str, Any] = {
|
|
"input_activations": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
"weights": {"dynamic": False, "num_bits": 8, "type": "float"},
|
|
"targets": ["Linear"],
|
|
}
|
|
new_config["config_groups"] = {"group_0": config_group_details}
|
|
elif quant_algo_value == "NVFP4":
|
|
group_size = original_quantization_details.get("group_size", 16)
|
|
config_group_details = {
|
|
"input_activations": {
|
|
"dynamic": False,
|
|
"num_bits": 4,
|
|
"type": "float",
|
|
"group_size": group_size,
|
|
},
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": group_size},
|
|
"targets": ["Linear"],
|
|
}
|
|
new_config["config_groups"] = {"group_0": config_group_details}
|
|
elif quant_algo_value == "W4A16_NVFP4":
|
|
# Weight-only FP4
|
|
group_size = original_quantization_details.get("group_size", 16)
|
|
config_group_details = {
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": group_size},
|
|
"targets": ["Linear"],
|
|
}
|
|
new_config["config_groups"] = {"group_0": config_group_details}
|
|
elif quant_algo_value in ("IQ1_S", "IQ2_XS"):
|
|
# Forward the caller's group size so a mismatched one is rejected rather than rewritten
|
|
# to the format's block size.
|
|
iq_metadata = _quant_algo_to_group_config(
|
|
quant_algo_value, original_quantization_details.get("group_size")
|
|
)
|
|
new_config.update(iq_metadata)
|
|
elif quant_algo_value == "NVFP4_SVD":
|
|
# NVFP4 + SVDQuant: NVFP4 weights/activations plus an AWQ-style
|
|
# pre_quant_scale and a low-rank residual (svdquant_lora_a/b) stored as
|
|
# <module>.pre_quant_scale / <module>.svdquant_lora_{a,b} in the
|
|
# safetensors. The config mirrors NVFP4 with a pre_quant_scale flag and
|
|
# the LoRA rank so consumers can reconstruct
|
|
# ``y = NVFP4_GEMM(x) + (x @ lora_a^T) @ lora_b^T``.
|
|
group_size = original_quantization_details.get("group_size", 16)
|
|
config_group_details = {
|
|
"input_activations": {
|
|
"dynamic": False,
|
|
"num_bits": 4,
|
|
"type": "float",
|
|
"group_size": group_size,
|
|
},
|
|
"weights": {"dynamic": False, "num_bits": 4, "type": "float", "group_size": group_size},
|
|
"has_zero_point": False,
|
|
"pre_quant_scale": True,
|
|
"targets": ["Linear"],
|
|
}
|
|
lora_rank = original_quantization_details.get("lora_rank")
|
|
if lora_rank is not None:
|
|
config_group_details["lora_rank"] = lora_rank
|
|
new_config["config_groups"] = {"group_0": config_group_details}
|
|
elif quant_algo_value == "MIXED_PRECISION":
|
|
quantized_layers = original_quantization_details.get("quantized_layers", {})
|
|
|
|
# Group layers by their unique quantization config so each distinct
|
|
# (quant_algo, group_size, ...) combination becomes one config_group.
|
|
algo_to_layers: dict[tuple, list[str]] = defaultdict(list)
|
|
for layer_name, layer_cfg in quantized_layers.items():
|
|
# Create a hashable key from the layer config
|
|
key = tuple(sorted(layer_cfg.items()))
|
|
algo_to_layers[key].append(layer_name)
|
|
|
|
config_groups: dict[str, Any] = {}
|
|
for idx, (config_key, layer_names) in enumerate(algo_to_layers.items()):
|
|
layer_cfg = dict(config_key)
|
|
algo = layer_cfg.get("quant_algo", "")
|
|
layer_group_size = layer_cfg.get("group_size")
|
|
|
|
group_config = _quant_algo_to_group_config(algo, layer_group_size)
|
|
group_config["targets"] = sorted(layer_names)
|
|
config_groups[f"group_{idx}"] = group_config
|
|
|
|
new_config["config_groups"] = config_groups
|
|
# Preserve the full per-layer detail for consumers that need it.
|
|
new_config["quantized_layers"] = quantized_layers
|
|
|
|
exclude_modules = original_quantization_details.get("exclude_modules")
|
|
|
|
new_config["ignore"] = exclude_modules if exclude_modules is not None else []
|
|
|
|
if quant_algo_value:
|
|
new_config["quant_algo"] = quant_algo_value
|
|
|
|
kv_cache_quant_algo = original_quantization_details.get("kv_cache_quant_algo")
|
|
if kv_cache_quant_algo:
|
|
if kv_cache_quant_algo == "FP8":
|
|
new_config["kv_cache_scheme"] = {"dynamic": False, "num_bits": 8, "type": "float"}
|
|
elif kv_cache_quant_algo in ("MIXED_PRECISION", "FP8_K_NVFP4_V"):
|
|
new_config["kv_cache_quant_algo"] = kv_cache_quant_algo
|
|
else:
|
|
# TODO: Handle other kv cache quantization algorithms
|
|
new_config["kv_cache_scheme"] = kv_cache_quant_algo
|
|
if "kv_cache_quantized_layers" in original_quantization_details:
|
|
new_config["kv_cache_quantized_layers"] = original_quantization_details[
|
|
"kv_cache_quantized_layers"
|
|
]
|
|
new_config["kv_cache_schema_version"] = original_quantization_details.get(
|
|
"kv_cache_schema_version", 1
|
|
)
|
|
|
|
producer_info = input_config.get("producer")
|
|
if producer_info:
|
|
new_config["producer"] = producer_info
|
|
|
|
new_config["quant_method"] = "modelopt"
|
|
|
|
return new_config
|