mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
Deprecate Mllama support in llm_ptq/vlm_ptq examples (#1332)
## Summary - Removes Mllama (Llama 3.2 Vision) model-type branches from the `llm_ptq` example (`hf_ptq.py`, `example_utils.py`) and drops the now-unused `MllamaImageProcessor` wrapper from `modelopt/torch/utils/`. - Drops the legacy `MllamaImageProcessor` path in `modelopt/torch/utils/vlm_dataset_utils.py`; the generic HF ProcessorMixin path handles the remaining cases. - Adds a CHANGELOG entry under 0.44 Backward Breaking Changes. ## Test plan - [x] CI lint / unit tests pass - [x] Smoke-run ``examples/llm_ptq/scripts/huggingface_example.sh --model <llm> --quant fp8`` (text-only path, non-mllama) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **Chores** * Removed Mllama (Llama 3.2 Vision) support from quantization examples. This includes removal of dedicated image processor implementation, specialized model handling, and related calibration logic. * Updated VLM image-text calibration guidance to use `--calib_with_images` flag with other supported VLMs instead of Mllama-specific processing paths. <!-- end of auto-generated comment: release notes by coderabbit.ai --> Signed-off-by: Chenjie Luo <chenjiel@nvidia.com>
This commit is contained in:
@@ -39,6 +39,7 @@ Changelog
|
||||
**Backward Breaking Changes**
|
||||
|
||||
- The ``quant_cfg`` field in quantization configs is now an **ordered list** of ``QuantizerCfgEntry`` dicts instead of a flat dictionary. Each entry specifies a ``quantizer_name`` wildcard, an optional ``parent_class`` filter, a ``cfg`` dict of quantizer attributes, and/or an ``enable`` flag. Entries are applied in list order with later entries overriding earlier ones. The old dict-based format is still accepted and automatically converted via ``normalize_quant_cfg_list()``, but now emits a ``DeprecationWarning``; new code should use the list format. All built-in configs (e.g. ``FP8_DEFAULT_CFG``, ``INT4_AWQ_CFG``, ``NVFP4_DEFAULT_CFG``), examples, and YAML recipes have been updated. See the :ref:`quant-cfg` documentation for the new format reference and migration guide.
|
||||
- Deprecated Mllama (Llama 3.2 Vision) support in the ``llm_ptq`` and ``vlm_ptq`` examples. The ``model_type == "mllama"`` branches and ``MllamaImageProcessor`` usage have been removed from ``hf_ptq.py`` and ``example_utils.py``. For image-text calibration of VLMs, use ``--calib_with_images`` with a supported VLM (see Nemotron VL section in ``examples/llm_ptq/README.md``).
|
||||
|
||||
**Bug Fixes**
|
||||
|
||||
|
||||
@@ -46,8 +46,6 @@ try:
|
||||
except ImportError:
|
||||
snapshot_download = None
|
||||
|
||||
from modelopt.torch.utils.image_processor import BaseImageProcessor, MllamaImageProcessor
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SPECULATIVE_MODEL_LIST = ["Eagle", "Medusa"]
|
||||
@@ -285,13 +283,10 @@ def get_tokenizer(ckpt_path, trust_remote_code=False, **kwargs) -> PreTrainedTok
|
||||
def get_processor(
|
||||
ckpt_path,
|
||||
model_type,
|
||||
device: torch.device = "auto",
|
||||
trust_remote_code=False,
|
||||
attn_implementation=None,
|
||||
) -> BaseImageProcessor | ProcessorMixin | None:
|
||||
"""
|
||||
Returns a :class:`modelopt.torch.utils.image_processor.MllamaImageProcessor` object.
|
||||
"""
|
||||
) -> ProcessorMixin | None:
|
||||
"""Load a processor appropriate for the given model type."""
|
||||
model_kwargs = {"trust_remote_code": trust_remote_code}
|
||||
if attn_implementation is not None:
|
||||
model_kwargs["attn_implementation"] = attn_implementation
|
||||
@@ -309,19 +304,6 @@ def get_processor(
|
||||
)
|
||||
|
||||
return processor
|
||||
elif model_type == "mllama":
|
||||
processor = AutoProcessor.from_pretrained(
|
||||
ckpt_path,
|
||||
padding_side="left",
|
||||
**model_kwargs,
|
||||
)
|
||||
if processor.tokenizer.pad_token is None:
|
||||
processor.tokenizer.pad_token = processor.tokenizer.eos_token
|
||||
assert processor.tokenizer.pad_token is not None, (
|
||||
f"Pad token for {ckpt_path} cannot be set!"
|
||||
)
|
||||
|
||||
return MllamaImageProcessor(processor, device)
|
||||
else:
|
||||
# Try to load AutoProcessor for other VL models (e.g., Nemotron-Parse)
|
||||
try:
|
||||
|
||||
@@ -77,7 +77,6 @@ from modelopt.torch.utils.dataset_utils import (
|
||||
get_max_batch_size,
|
||||
get_supported_datasets,
|
||||
)
|
||||
from modelopt.torch.utils.image_processor import BaseImageProcessor, MllamaImageProcessor
|
||||
from modelopt.torch.utils.memory_monitor import launch_memory_monitor
|
||||
from modelopt.torch.utils.speech_dataset_utils import get_speech_dataset_dataloader
|
||||
from modelopt.torch.utils.vlm_dataset_utils import get_vlm_dataset_dataloader
|
||||
@@ -202,7 +201,7 @@ def _move_batch_to_device(batch: dict, device: torch.device) -> dict:
|
||||
def make_calib_dataloader(
|
||||
args: argparse.Namespace,
|
||||
language_model: torch.nn.Module,
|
||||
processor: BaseImageProcessor | ProcessorMixin | None,
|
||||
processor: ProcessorMixin | None,
|
||||
tokenizer: PreTrainedTokenizerBase | None,
|
||||
device: torch.device,
|
||||
model_type: str | None,
|
||||
@@ -250,19 +249,6 @@ def make_calib_dataloader(
|
||||
use_media_shards=True,
|
||||
max_shards=1,
|
||||
)
|
||||
elif model_type == "mllama":
|
||||
assert processor is not None and isinstance(processor, MllamaImageProcessor), (
|
||||
"The MllamaImageProcessor must be set."
|
||||
)
|
||||
assert len(args.calib_size) == 1, (
|
||||
"mllama only supports one dataset for calibration, can extend this in the future"
|
||||
)
|
||||
calib_dataloader = get_vlm_dataset_dataloader(
|
||||
dataset_name=args.dataset[0] if args.dataset else "scienceqa",
|
||||
processor=processor,
|
||||
batch_size=args.batch_size,
|
||||
num_samples=args.calib_size[0],
|
||||
)
|
||||
elif model_type == "whisper":
|
||||
assert processor is not None and isinstance(processor, WhisperProcessor), (
|
||||
"The AutoProcessor must be set."
|
||||
@@ -473,19 +459,10 @@ def load_model(args: argparse.Namespace):
|
||||
print("Nemotron VL model detected. Enabling image-text calibration by default.")
|
||||
args.calib_with_images = True
|
||||
|
||||
if model_type == "mllama":
|
||||
if model_type == "whisper":
|
||||
processor = get_processor(
|
||||
args.pyt_ckpt_path,
|
||||
model_type,
|
||||
device,
|
||||
trust_remote_code=args.trust_remote_code,
|
||||
attn_implementation=args.attn_implementation,
|
||||
)
|
||||
elif model_type == "whisper":
|
||||
processor = get_processor(
|
||||
args.pyt_ckpt_path,
|
||||
model_type,
|
||||
device,
|
||||
trust_remote_code=args.trust_remote_code,
|
||||
)
|
||||
elif is_nemotron_vl_model and args.calib_with_images:
|
||||
@@ -716,13 +693,6 @@ def export_quantized(
|
||||
print(f"Warning: Could not save processor config: {e}")
|
||||
print("This is normal for some VLM architectures that don't use AutoProcessor")
|
||||
|
||||
if model_type == "mllama":
|
||||
full_model_config = full_model.config
|
||||
# TRT-LLM expects both the vision_config and text_config to be set for export.
|
||||
setattr(full_model.config, "vision_config", full_model_config.vision_config)
|
||||
setattr(full_model.config, "text_config", full_model_config.text_config)
|
||||
setattr(full_model.config, "architectures", full_model_config.architectures)
|
||||
|
||||
start_time = time.time()
|
||||
if (
|
||||
model_type in ["t5", "bart", "whisper"]
|
||||
@@ -859,7 +829,7 @@ def post_quantize(
|
||||
language_model: torch.nn.Module,
|
||||
model_type: str | None,
|
||||
tokenizer: PreTrainedTokenizerBase | None,
|
||||
processor: BaseImageProcessor | ProcessorMixin | None,
|
||||
processor: ProcessorMixin | None,
|
||||
preview_input_ids,
|
||||
generated_ids_before_ptq,
|
||||
is_nemotron_vl_model,
|
||||
@@ -922,9 +892,7 @@ def post_quantize(
|
||||
)
|
||||
|
||||
def input_decode(input_ids):
|
||||
if processor is not None and isinstance(processor, MllamaImageProcessor):
|
||||
return processor.tokenizer.batch_decode(input_ids)
|
||||
elif processor is not None and isinstance(processor, WhisperProcessor):
|
||||
if processor is not None and isinstance(processor, WhisperProcessor):
|
||||
return first_text_speech_dataset
|
||||
elif tokenizer is not None:
|
||||
return tokenizer.batch_decode(input_ids)
|
||||
@@ -937,8 +905,6 @@ def post_quantize(
|
||||
return processor.tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
|
||||
elif tokenizer is not None:
|
||||
return tokenizer.batch_decode(generated_ids, skip_special_tokens=True)
|
||||
elif processor is not None and isinstance(processor, MllamaImageProcessor):
|
||||
return processor.tokenizer.batch_decode(generated_ids[:, input_shape:])
|
||||
elif tokenizer is not None:
|
||||
return tokenizer.batch_decode(generated_ids[:, input_shape:])
|
||||
else:
|
||||
@@ -983,7 +949,7 @@ def quantize_main(
|
||||
language_model: torch.nn.Module,
|
||||
model_type: str | None,
|
||||
calibration_only: bool,
|
||||
processor: BaseImageProcessor | ProcessorMixin | None,
|
||||
processor: ProcessorMixin | None,
|
||||
tokenizer: PreTrainedTokenizerBase | None,
|
||||
default_padding_side,
|
||||
default_pad_token,
|
||||
|
||||
@@ -1,112 +0,0 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# Adapted from tensorrt_llm/quantization/image_processing.py
|
||||
"""Utility classes for image processing."""
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
class BaseImageProcessor:
|
||||
"""Base class for image processors."""
|
||||
|
||||
def __init__(self, tokenizer, device="cuda"):
|
||||
"""Constructor."""
|
||||
self.tokenizer = tokenizer
|
||||
self.device = device
|
||||
|
||||
def __call__(self, **kwargs):
|
||||
"""Call the tokenizer."""
|
||||
return self.tokenizer(**kwargs)
|
||||
|
||||
def preprocess_function(self, examples):
|
||||
"""Preprocess function."""
|
||||
raise NotImplementedError("Each image processor must implement its own preprocess method")
|
||||
|
||||
def collate_function(self, examples):
|
||||
"""Collate function to process images during data loading."""
|
||||
raise NotImplementedError("Each image processor must implement its own collate method")
|
||||
|
||||
|
||||
# A light Encapsulation for Huggingface MllamaImageProcessor
|
||||
|
||||
|
||||
class MllamaImageProcessor(BaseImageProcessor):
|
||||
"""Image processor for Mllama."""
|
||||
|
||||
def preprocess_function(self, examples):
|
||||
"""Preprocess function."""
|
||||
# Prepare prompts in a generic chat format
|
||||
question = examples.get("question", "Describe this image.")
|
||||
|
||||
if examples["image"] is not None:
|
||||
if self.tokenizer.chat_template is not None:
|
||||
prompt = self.tokenizer.apply_chat_template(
|
||||
[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"type": "image"}, {"type": "text", "text": question}],
|
||||
}
|
||||
],
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
else:
|
||||
prompt = f"<|image|><|begin_of_text|>{question}"
|
||||
|
||||
# Process images using the processor's image processor
|
||||
values = self.tokenizer(text=prompt, images=examples["image"], return_tensors="pt").to(
|
||||
self.device
|
||||
)
|
||||
else:
|
||||
if self.tokenizer.chat_template is not None:
|
||||
prompt = self.tokenizer.apply_chat_template(
|
||||
[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"type": "text", "text": question}],
|
||||
}
|
||||
],
|
||||
add_generation_prompt=True,
|
||||
)
|
||||
else:
|
||||
prompt = question
|
||||
|
||||
values = self.tokenizer(text=prompt, images=None, return_tensors="pt").to(self.device)
|
||||
|
||||
values["pixel_values"] = None
|
||||
values["aspect_ratio_ids"] = None
|
||||
values["aspect_ratio_mask"] = None
|
||||
values["cross_attention_mask"] = None
|
||||
|
||||
return values
|
||||
|
||||
def collate_function(self, batch):
|
||||
"""Collate function to process images during data loading."""
|
||||
batch[0]["input_ids"] = torch.LongTensor(batch[0]["input_ids"]).to(self.device)
|
||||
batch[0]["attention_mask"] = torch.LongTensor(batch[0]["attention_mask"]).to(self.device)
|
||||
|
||||
if batch[0]["pixel_values"] is not None:
|
||||
batch[0]["pixel_values"] = torch.Tensor(batch[0]["pixel_values"]).to(self.device)
|
||||
batch[0]["aspect_ratio_ids"] = torch.LongTensor(batch[0]["aspect_ratio_ids"]).to(
|
||||
self.device
|
||||
)
|
||||
batch[0]["aspect_ratio_mask"] = torch.LongTensor(batch[0]["aspect_ratio_mask"]).to(
|
||||
self.device
|
||||
)
|
||||
batch[0]["cross_attention_mask"] = torch.LongTensor(
|
||||
batch[0]["cross_attention_mask"]
|
||||
).to(self.device)
|
||||
|
||||
return batch[0]
|
||||
@@ -30,7 +30,6 @@ from typing import Any
|
||||
import torch
|
||||
from torch.utils.data import DataLoader
|
||||
|
||||
from .image_processor import MllamaImageProcessor
|
||||
from .nemotron_vlm_dataset_utils import NemotronTarPlusJsonlIterable, list_repo_files_cached
|
||||
|
||||
# Use dict to store the config for each dataset.
|
||||
@@ -379,18 +378,6 @@ def get_vlm_dataset_dataloader(
|
||||
max_shards=max_shards,
|
||||
)
|
||||
|
||||
# Legacy path: our internal image processor wrapper (e.g., Mllama).
|
||||
if isinstance(processor, MllamaImageProcessor):
|
||||
processed_dataset = dataset.map(
|
||||
processor.preprocess_function, batched=False, remove_columns=dataset.column_names
|
||||
)
|
||||
return DataLoader(
|
||||
processed_dataset,
|
||||
batch_size=batch_size,
|
||||
shuffle=False,
|
||||
collate_fn=processor.collate_function,
|
||||
)
|
||||
|
||||
# Generic HF ProcessorMixin / AutoProcessor path: tokenize & process images at collate-time.
|
||||
# For Nemotron VLM datasets, we prefer to follow the model-card flow:
|
||||
# prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
|
||||
|
||||
Reference in New Issue
Block a user