mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? - Add experimental support for transformers >=5.0 and remove deprecated usages: https://github.com/huggingface/transformers/blob/main/MIGRATION_GUIDE_V5.md - ⚠️ For accelerate examples that used `--warmup-ratio: float` (deprecated in 5.x), we now change it to `--warmup-steps: float | int` which works as ratio if float but only for 5.x. For 4.x, it will error out if float and prompt user to change back to `--warmup-ratio` or pass an int absolute step count. - ⚠️ Unified Hugging Face checkpoint export for quantized checkpoints may not work for some models with transformers>=5.0 yet as it requires a lot of fixes (e.g. change in how MoE experts are organized) - ~Add Workaround for TRT-LLM's import of deprecated transformers functions so trt-llm based gpu unit tests work fine. Still deployment for models needs proper fixes directly in TRT-LLM hence llm/vlm ptq example tests still run with transformers 4.57~ - Everything except PTQ and Export (mainly MoE) should work fine with transformers>=5.0 - Bump min torch to 2.8 and enable 2.11 cicd testing - NOTE: Upcoming Nemo:26.04 container comes with transformers 5.3 ### Testing <!-- Mention how have you tested your change if applicable. --> - [x] CI/CD tests passing - [x] Manually tested unit tests, gpu tests with transformers 4.56 and 5.4 - [x] Manually tested example tests (except trt-llm container tests) with transformers 4.56 and 5.4 - [x] 2-gpu nightly CICD tests manually triggered and passing: [gpu tests](https://github.com/NVIDIA/Model-Optimizer/actions/runs/23867257540), [example tests](https://github.com/NVIDIA/Model-Optimizer/actions/runs/23867260643) ### Before your PR is "*Ready for review*" Make sure you read and follow [Contributor guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md) and your commits are signed (`git commit -s -S`). Make sure you read and follow the [Security Best Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors) (e.g. avoiding hardcoded `trust_remote_code=True`, using `torch.load(..., weights_only=True)`, avoiding `pickle`, etc.). - Is this change backward compatible?: ✅ <!--- If ❌, explain why. --> - If you copied code from any other source, did you follow IP policy in [CONTRIBUTING.md](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md#-copying-code-from-other-sources)?: N/A <!--- Mandatory --> - Did you write any new necessary tests?: ✅ <!--- Mandatory for new features or examples. --> - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: ✅ <!--- Only for new features, API changes, critical bug fixes or backward incompatible changes. --> <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Make remote-code usage opt-in via a configurable --trust_remote_code flag across examples and tools. * **Bug Fixes** * Improve checkpoint/resume detection and related training guidance to avoid erroneous errors. * **Refactor** * Consolidate dtype/config naming, switch warmup settings from ratio → steps, and unify tokenizer invocation patterns. * **Documentation** * Simplify changelog title and add misc notes for release 0.44. * **Chores** * Remove scheduled PR-branch cleanup workflow and relax/remove several transformers version pins. * **Tests** * Adjust test gates, skips, and structures to align with updated deps and behaviors. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com>
276 lines
9.2 KiB
Python
276 lines
9.2 KiB
Python
#!/usr/bin/env python3
|
|
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""Example script for applying sparse attention to HuggingFace models."""
|
|
|
|
import argparse
|
|
import copy
|
|
import random
|
|
from pathlib import Path
|
|
|
|
import numpy as np
|
|
import torch
|
|
from transformers import AutoModelForCausalLM, AutoTokenizer
|
|
|
|
import modelopt.torch.opt as mto
|
|
import modelopt.torch.sparsity.attention_sparsity as mtsa
|
|
from modelopt.torch.export import export_hf_checkpoint
|
|
from modelopt.torch.sparsity.attention_sparsity.config import SKIP_SOFTMAX_CALIB
|
|
from modelopt.torch.utils.memory_monitor import launch_memory_monitor
|
|
|
|
RAND_SEED = 1234
|
|
|
|
# Enable HuggingFace checkpointing support
|
|
mto.enable_huggingface_checkpointing()
|
|
|
|
# Sparse attention configuration choices
|
|
SPARSE_ATTN_CFG_CHOICES = {
|
|
"skip_softmax_calib": SKIP_SOFTMAX_CALIB,
|
|
}
|
|
|
|
|
|
def get_test_prompts():
|
|
"""Get simple test prompts for sample output generation."""
|
|
return [
|
|
"What is the capital of France? Answer:",
|
|
"Explain the theory of relativity in simple terms:",
|
|
"Write a short poem about the ocean:",
|
|
]
|
|
|
|
|
|
def truncate_text(text: str, tokenizer, max_length: int):
|
|
"""Truncate text from the middle to preserve beginning and end.
|
|
|
|
Args:
|
|
text: Input text to truncate
|
|
tokenizer: Tokenizer to use for encoding
|
|
max_length: Maximum number of tokens
|
|
|
|
Returns:
|
|
Truncated text that fits within max_length tokens
|
|
"""
|
|
# First tokenize to see if truncation is needed
|
|
tokens = tokenizer.encode(text, add_special_tokens=True)
|
|
|
|
if len(tokens) <= max_length:
|
|
return text
|
|
|
|
# Need to truncate - preserve beginning and end
|
|
# Calculate actual special tokens used
|
|
dummy_tokens = tokenizer.encode("", add_special_tokens=True)
|
|
special_token_count = len(dummy_tokens)
|
|
available_tokens = max_length - special_token_count
|
|
|
|
# Split tokens roughly in half for beginning and end
|
|
begin_tokens = available_tokens // 2
|
|
end_tokens = available_tokens - begin_tokens
|
|
|
|
# Decode beginning and end parts
|
|
begin_text = tokenizer.decode(tokens[:begin_tokens], skip_special_tokens=True)
|
|
end_text = tokenizer.decode(tokens[-end_tokens:], skip_special_tokens=True)
|
|
|
|
# Combine with ellipsis marker
|
|
return begin_text + " [...] " + end_text
|
|
|
|
|
|
def generate_sample_output(model, tokenizer, args):
|
|
"""Generate sample output for comparison.
|
|
|
|
Args:
|
|
model: The model to generate with
|
|
tokenizer: Tokenizer for encoding/decoding
|
|
args: Command line arguments
|
|
|
|
Returns:
|
|
Tuple of (generated_text, input_prompt, input_ids)
|
|
"""
|
|
# Load test sample
|
|
prompts = get_test_prompts()
|
|
prompt = prompts[0]
|
|
|
|
# Prepare inputs
|
|
truncated_prompt = truncate_text(prompt, tokenizer, args.seq_len)
|
|
inputs = tokenizer(
|
|
truncated_prompt,
|
|
return_tensors="pt",
|
|
max_length=args.seq_len,
|
|
truncation=True,
|
|
padding=False,
|
|
)
|
|
if torch.cuda.is_available():
|
|
inputs = {k: v.to(model.device) for k, v in inputs.items()}
|
|
|
|
# Generate
|
|
with torch.no_grad():
|
|
outputs = model.generate(
|
|
**inputs,
|
|
max_new_tokens=args.max_new_tokens,
|
|
do_sample=args.do_sample,
|
|
temperature=args.temperature if args.do_sample else 1.0,
|
|
pad_token_id=tokenizer.pad_token_id,
|
|
)
|
|
input_length = inputs["input_ids"].shape[1]
|
|
generated_ids = outputs[0][input_length:]
|
|
generated_text = tokenizer.decode(generated_ids, skip_special_tokens=True)
|
|
|
|
return generated_text, truncated_prompt, inputs["input_ids"]
|
|
|
|
|
|
def main(args):
|
|
"""Main function to run the selected mode."""
|
|
if not torch.cuda.is_available():
|
|
raise OSError("GPU is required for inference.")
|
|
|
|
random.seed(RAND_SEED)
|
|
np.random.seed(RAND_SEED)
|
|
launch_memory_monitor()
|
|
|
|
print(f"Loading model: {args.pyt_ckpt_path}")
|
|
|
|
# No need to specify attn_implementation here — mtsa.sparsify() sets it
|
|
# automatically ("eager" for pytorch backend, "modelopt_triton" for triton).
|
|
model = AutoModelForCausalLM.from_pretrained(
|
|
args.pyt_ckpt_path, attn_implementation="eager", dtype="auto", device_map="auto"
|
|
)
|
|
tokenizer = AutoTokenizer.from_pretrained(args.pyt_ckpt_path)
|
|
|
|
# Set pad token if not set
|
|
if tokenizer.pad_token is None:
|
|
tokenizer.pad_token = tokenizer.eos_token
|
|
|
|
# Generate sample output BEFORE sparse attention
|
|
print("\nGenerating sample output before sparse attention...")
|
|
output_before, test_prompt, input_ids = generate_sample_output(model, tokenizer, args)
|
|
|
|
# Apply sparse attention with optional calibration
|
|
print(f"\nApplying sparse attention: {args.sparse_attn} (backend={args.backend})")
|
|
sparse_config = copy.deepcopy(SPARSE_ATTN_CFG_CHOICES[args.sparse_attn])
|
|
|
|
# Apply CLI overrides to sparse_cfg
|
|
sparse_cfg = sparse_config.get("sparse_cfg", {})
|
|
if args.backend is not None:
|
|
for layer_cfg in sparse_cfg.values():
|
|
if isinstance(layer_cfg, dict) and "method" in layer_cfg:
|
|
layer_cfg["backend"] = args.backend
|
|
if args.target_sparse_ratio is not None:
|
|
calib = sparse_cfg.setdefault("calibration", {})
|
|
assert isinstance(calib, dict)
|
|
calib["target_sparse_ratio"] = {
|
|
"prefill": args.target_sparse_ratio,
|
|
"decode": args.target_sparse_ratio,
|
|
}
|
|
|
|
model = mtsa.sparsify(model, config=sparse_config)
|
|
print("Sparse attention applied successfully!")
|
|
|
|
# Generate sample output AFTER sparse attention
|
|
print("\nGenerating sample output after sparse attention...")
|
|
output_after, _, _ = generate_sample_output(model, tokenizer, args)
|
|
|
|
# Display comparison
|
|
print("\n" + "=" * 60)
|
|
print("OUTPUT COMPARISON (Before vs After Sparse Attention)")
|
|
print("=" * 60)
|
|
display_prompt = test_prompt[:150] + "..." if len(test_prompt) > 150 else test_prompt
|
|
print(f"\nTest prompt: {display_prompt}")
|
|
print(f"Input tokens: {input_ids.shape[1]}")
|
|
|
|
output_before_display = (
|
|
output_before[:300] + "..." if len(output_before) > 300 else output_before
|
|
)
|
|
output_after_display = output_after[:300] + "..." if len(output_after) > 300 else output_after
|
|
|
|
print(f"\nBefore sparse attention: {output_before_display}")
|
|
print(f"After sparse attention: {output_after_display}")
|
|
|
|
if output_before == output_after:
|
|
print("\nOutputs are identical")
|
|
else:
|
|
print("\nOutputs differ")
|
|
|
|
# Export if requested
|
|
if args.export_dir:
|
|
print(f"\nExporting model to: {args.export_dir}")
|
|
export_dir = Path(args.export_dir)
|
|
export_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
with torch.inference_mode():
|
|
export_hf_checkpoint(model, export_dir=export_dir)
|
|
|
|
tokenizer.save_pretrained(export_dir)
|
|
print(f"Model exported successfully to: {export_dir}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
|
|
# Model arguments
|
|
parser.add_argument(
|
|
"--pyt_ckpt_path",
|
|
type=str,
|
|
required=True,
|
|
help="Specify where the PyTorch checkpoint path is",
|
|
)
|
|
parser.add_argument(
|
|
"--sparse_attn",
|
|
type=str,
|
|
default="skip_softmax_calib",
|
|
choices=list(SPARSE_ATTN_CFG_CHOICES.keys()),
|
|
help="Sparse attention configuration to apply.",
|
|
)
|
|
parser.add_argument(
|
|
"--backend",
|
|
type=str,
|
|
default=None,
|
|
choices=["pytorch", "triton"],
|
|
help="Backend for sparse attention. Overrides the config default if set. "
|
|
"'triton' uses the fused Triton kernel.",
|
|
)
|
|
|
|
# Sequence length arguments
|
|
parser.add_argument(
|
|
"--seq_len",
|
|
type=int,
|
|
default=2048,
|
|
help="Maximum sequence length for input prompts (will be truncated if longer)",
|
|
)
|
|
|
|
# Generation arguments
|
|
parser.add_argument(
|
|
"--max_new_tokens", type=int, default=50, help="Maximum new tokens to generate"
|
|
)
|
|
parser.add_argument("--do_sample", action="store_true", help="Use sampling for generation")
|
|
parser.add_argument("--temperature", type=float, default=0.7, help="Temperature for sampling")
|
|
|
|
# Operation arguments
|
|
parser.add_argument(
|
|
"--export_dir",
|
|
type=str,
|
|
default=None,
|
|
help="Directory to export the model with sparse attention applied",
|
|
)
|
|
|
|
# Calibration arguments
|
|
parser.add_argument(
|
|
"--target_sparse_ratio",
|
|
type=float,
|
|
default=None,
|
|
help="Target sparsity ratio for calibration (0.0 to 1.0). Overrides config value.",
|
|
)
|
|
|
|
args = parser.parse_args()
|
|
main(args)
|