Deprecate Mllama support in llm_ptq/vlm_ptq examples (#1332)

## Summary

- Removes Mllama (Llama 3.2 Vision) model-type branches from the
`llm_ptq` example (`hf_ptq.py`, `example_utils.py`) and drops the
now-unused `MllamaImageProcessor` wrapper from `modelopt/torch/utils/`.
- Drops the legacy `MllamaImageProcessor` path in
`modelopt/torch/utils/vlm_dataset_utils.py`; the generic HF
ProcessorMixin path handles the remaining cases.
- Adds a CHANGELOG entry under 0.44 Backward Breaking Changes.

## Test plan

- [x] CI lint / unit tests pass
- [x] Smoke-run ``examples/llm_ptq/scripts/huggingface_example.sh
--model <llm> --quant fp8`` (text-only path, non-mllama)

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->

## Summary by CodeRabbit

* **Chores**
* Removed Mllama (Llama 3.2 Vision) support from quantization examples.
This includes removal of dedicated image processor implementation,
specialized model handling, and related calibration logic.
* Updated VLM image-text calibration guidance to use
`--calib_with_images` flag with other supported VLMs instead of
Mllama-specific processing paths.

<!-- end of auto-generated comment: release notes by coderabbit.ai -->

Signed-off-by: Chenjie Luo <chenjiel@nvidia.com>
This commit is contained in:
Chenjie Luo
2026-04-23 23:24:42 +05:30
committed by GitHub
parent 8663678f12
commit 01788bb007
5 changed files with 8 additions and 184 deletions
+1
View File
@@ -39,6 +39,7 @@ Changelog
**Backward Breaking Changes**
- The ``quant_cfg`` field in quantization configs is now an **ordered list** of ``QuantizerCfgEntry`` dicts instead of a flat dictionary. Each entry specifies a ``quantizer_name`` wildcard, an optional ``parent_class`` filter, a ``cfg`` dict of quantizer attributes, and/or an ``enable`` flag. Entries are applied in list order with later entries overriding earlier ones. The old dict-based format is still accepted and automatically converted via ``normalize_quant_cfg_list()``, but now emits a ``DeprecationWarning``; new code should use the list format. All built-in configs (e.g. ``FP8_DEFAULT_CFG``, ``INT4_AWQ_CFG``, ``NVFP4_DEFAULT_CFG``), examples, and YAML recipes have been updated. See the :ref:`quant-cfg` documentation for the new format reference and migration guide.
- Deprecated Mllama (Llama 3.2 Vision) support in the ``llm_ptq`` and ``vlm_ptq`` examples. The ``model_type == "mllama"`` branches and ``MllamaImageProcessor`` usage have been removed from ``hf_ptq.py`` and ``example_utils.py``. For image-text calibration of VLMs, use ``--calib_with_images`` with a supported VLM (see Nemotron VL section in ``examples/llm_ptq/README.md``).
**Bug Fixes**
+2 -20
View File
@@ -46,8 +46,6 @@ try:
except ImportError:
snapshot_download = None
from modelopt.torch.utils.image_processor import BaseImageProcessor, MllamaImageProcessor
logger = logging.getLogger(__name__)
SPECULATIVE_MODEL_LIST = ["Eagle", "Medusa"]
@@ -285,13 +283,10 @@ def get_tokenizer(ckpt_path, trust_remote_code=False, **kwargs) -> PreTrainedTok
def get_processor(
ckpt_path,
model_type,
device: torch.device = "auto",
trust_remote_code=False,
attn_implementation=None,
) -> BaseImageProcessor | ProcessorMixin | None:
"""
Returns a :class:`modelopt.torch.utils.image_processor.MllamaImageProcessor` object.
"""
) -> ProcessorMixin | None:
"""Load a processor appropriate for the given model type."""
model_kwargs = {"trust_remote_code": trust_remote_code}
if attn_implementation is not None:
model_kwargs["attn_implementation"] = attn_implementation
@@ -309,19 +304,6 @@ def get_processor(
)
return processor
elif model_type == "mllama":
processor = AutoProcessor.from_pretrained(
ckpt_path,
padding_side="left",
**model_kwargs,
)
if processor.tokenizer.pad_token is None:
processor.tokenizer.pad_token = processor.tokenizer.eos_token
assert processor.tokenizer.pad_token is not None, (
f"Pad token for {ckpt_path} cannot be set!"
)
return MllamaImageProcessor(processor, device)
else:
# Try to load AutoProcessor for other VL models (e.g., Nemotron-Parse)
try:
+5 -39
View File
@@ -77,7 +77,6 @@ from modelopt.torch.utils.dataset_utils import (
get_max_batch_size,
get_supported_datasets,
)
from modelopt.torch.utils.image_processor import BaseImageProcessor, MllamaImageProcessor
from modelopt.torch.utils.memory_monitor import launch_memory_monitor
from modelopt.torch.utils.speech_dataset_utils import get_speech_dataset_dataloader
from modelopt.torch.utils.vlm_dataset_utils import get_vlm_dataset_dataloader
@@ -202,7 +201,7 @@ def _move_batch_to_device(batch: dict, device: torch.device) -> dict:
def make_calib_dataloader(
args: argparse.Namespace,
language_model: torch.nn.Module,
processor: BaseImageProcessor | ProcessorMixin | None,
processor: ProcessorMixin | None,
tokenizer: PreTrainedTokenizerBase | None,
device: torch.device,
model_type: str | None,
@@ -250,19 +249,6 @@ def make_calib_dataloader(
use_media_shards=True,
max_shards=1,
)
elif model_type == "mllama":
assert processor is not None and isinstance(processor, MllamaImageProcessor), (
"The MllamaImageProcessor must be set."
)
assert len(args.calib_size) == 1, (
"mllama only supports one dataset for calibration, can extend this in the future"
)
calib_dataloader = get_vlm_dataset_dataloader(
dataset_name=args.dataset[0] if args.dataset else "scienceqa",
processor=processor,
batch_size=args.batch_size,
num_samples=args.calib_size[0],
)
elif model_type == "whisper":
assert processor is not None and isinstance(processor, WhisperProcessor), (
"The AutoProcessor must be set."
@@ -473,19 +459,10 @@ def load_model(args: argparse.Namespace):
print("Nemotron VL model detected. Enabling image-text calibration by default.")
args.calib_with_images = True
if model_type == "mllama":
if model_type == "whisper":
processor = get_processor(
args.pyt_ckpt_path,
model_type,
device,
trust_remote_code=args.trust_remote_code,
attn_implementation=args.attn_implementation,
)
elif model_type == "whisper":
processor = get_processor(
args.pyt_ckpt_path,
model_type,
device,
trust_remote_code=args.trust_remote_code,
)
elif is_nemotron_vl_model and args.calib_with_images:
@@ -716,13 +693,6 @@ def export_quantized(
print(f"Warning: Could not save processor config: {e}")
print("This is normal for some VLM architectures that don't use AutoProcessor")
if model_type == "mllama":
full_model_config = full_model.config
# TRT-LLM expects both the vision_config and text_config to be set for export.
setattr(full_model.config, "vision_config", full_model_config.vision_config)
setattr(full_model.config, "text_config", full_model_config.text_config)
setattr(full_model.config, "architectures", full_model_config.architectures)
start_time = time.time()
if (
model_type in ["t5", "bart", "whisper"]
@@ -859,7 +829,7 @@ def post_quantize(
language_model: torch.nn.Module,
model_type: str | None,
tokenizer: PreTrainedTokenizerBase | None,
processor: BaseImageProcessor | ProcessorMixin | None,
processor: ProcessorMixin | None,
preview_input_ids,
generated_ids_before_ptq,
is_nemotron_vl_model,
@@ -922,9 +892,7 @@ def post_quantize(
)
def input_decode(input_ids):
if processor is not None and isinstance(processor, MllamaImageProcessor):
return processor.tokenizer.batch_decode(input_ids)
elif processor is not None and isinstance(processor, WhisperProcessor):
if processor is not None and isinstance(processor, WhisperProcessor):
return first_text_speech_dataset
elif tokenizer is not None:
return tokenizer.batch_decode(input_ids)
@@ -937,8 +905,6 @@ def post_quantize(
return processor.tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
elif tokenizer is not None:
return tokenizer.batch_decode(generated_ids, skip_special_tokens=True)
elif processor is not None and isinstance(processor, MllamaImageProcessor):
return processor.tokenizer.batch_decode(generated_ids[:, input_shape:])
elif tokenizer is not None:
return tokenizer.batch_decode(generated_ids[:, input_shape:])
else:
@@ -983,7 +949,7 @@ def quantize_main(
language_model: torch.nn.Module,
model_type: str | None,
calibration_only: bool,
processor: BaseImageProcessor | ProcessorMixin | None,
processor: ProcessorMixin | None,
tokenizer: PreTrainedTokenizerBase | None,
default_padding_side,
default_pad_token,
-112
View File
@@ -1,112 +0,0 @@
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Adapted from tensorrt_llm/quantization/image_processing.py
"""Utility classes for image processing."""
import torch
class BaseImageProcessor:
"""Base class for image processors."""
def __init__(self, tokenizer, device="cuda"):
"""Constructor."""
self.tokenizer = tokenizer
self.device = device
def __call__(self, **kwargs):
"""Call the tokenizer."""
return self.tokenizer(**kwargs)
def preprocess_function(self, examples):
"""Preprocess function."""
raise NotImplementedError("Each image processor must implement its own preprocess method")
def collate_function(self, examples):
"""Collate function to process images during data loading."""
raise NotImplementedError("Each image processor must implement its own collate method")
# A light Encapsulation for Huggingface MllamaImageProcessor
class MllamaImageProcessor(BaseImageProcessor):
"""Image processor for Mllama."""
def preprocess_function(self, examples):
"""Preprocess function."""
# Prepare prompts in a generic chat format
question = examples.get("question", "Describe this image.")
if examples["image"] is not None:
if self.tokenizer.chat_template is not None:
prompt = self.tokenizer.apply_chat_template(
[
{
"role": "user",
"content": [{"type": "image"}, {"type": "text", "text": question}],
}
],
add_generation_prompt=True,
)
else:
prompt = f"<|image|><|begin_of_text|>{question}"
# Process images using the processor's image processor
values = self.tokenizer(text=prompt, images=examples["image"], return_tensors="pt").to(
self.device
)
else:
if self.tokenizer.chat_template is not None:
prompt = self.tokenizer.apply_chat_template(
[
{
"role": "user",
"content": [{"type": "text", "text": question}],
}
],
add_generation_prompt=True,
)
else:
prompt = question
values = self.tokenizer(text=prompt, images=None, return_tensors="pt").to(self.device)
values["pixel_values"] = None
values["aspect_ratio_ids"] = None
values["aspect_ratio_mask"] = None
values["cross_attention_mask"] = None
return values
def collate_function(self, batch):
"""Collate function to process images during data loading."""
batch[0]["input_ids"] = torch.LongTensor(batch[0]["input_ids"]).to(self.device)
batch[0]["attention_mask"] = torch.LongTensor(batch[0]["attention_mask"]).to(self.device)
if batch[0]["pixel_values"] is not None:
batch[0]["pixel_values"] = torch.Tensor(batch[0]["pixel_values"]).to(self.device)
batch[0]["aspect_ratio_ids"] = torch.LongTensor(batch[0]["aspect_ratio_ids"]).to(
self.device
)
batch[0]["aspect_ratio_mask"] = torch.LongTensor(batch[0]["aspect_ratio_mask"]).to(
self.device
)
batch[0]["cross_attention_mask"] = torch.LongTensor(
batch[0]["cross_attention_mask"]
).to(self.device)
return batch[0]
-13
View File
@@ -30,7 +30,6 @@ from typing import Any
import torch
from torch.utils.data import DataLoader
from .image_processor import MllamaImageProcessor
from .nemotron_vlm_dataset_utils import NemotronTarPlusJsonlIterable, list_repo_files_cached
# Use dict to store the config for each dataset.
@@ -379,18 +378,6 @@ def get_vlm_dataset_dataloader(
max_shards=max_shards,
)
# Legacy path: our internal image processor wrapper (e.g., Mllama).
if isinstance(processor, MllamaImageProcessor):
processed_dataset = dataset.map(
processor.preprocess_function, batched=False, remove_columns=dataset.column_names
)
return DataLoader(
processed_dataset,
batch_size=batch_size,
shuffle=False,
collate_fn=processor.collate_function,
)
# Generic HF ProcessorMixin / AutoProcessor path: tokenize & process images at collate-time.
# For Nemotron VLM datasets, we prefer to follow the model-card flow:
# prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)