mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
## What does this PR do?
**Type of change:** ? <!-- Use one of the following: Bug fix, new
feature, new example, new tests, documentation. -->
new feature
**Overview:** ?
- This PR adds the sparse attention calibration algorithm
- Chunked prefill to support long ctx_len
- Separated calibration for prefill and decode
## Usage
<!-- You can potentially add a usage example below. -->
```python
import modelopt.torch.sparsity.attention_sparsity as mtsa
# Apply sparse attention with calibration
model = mtsa.sparsify(model, config=SKIP_SOFTMAX_CALIB)
# Print summary - now shows actual thresholds
mtsa.print_sparse_attention_summary(model)
# Output:
# Method: flash_skip_softmax, Threshold: Dynamic (λ=437.395926)
# Or llm_eval integration
# HuggingFace sparse attention example
python examples/llm_sparsity/attention_sparsity/hf_sa.py \
--pyt_ckpt_path Qwen/Qwen3-4B \
--sparse_attn skip_softmax_calib
```
# The calibration method
## Calibration Algorithm
- Implemented the Inverse Power model: scale_factor = k / (1 -
sparsity)^p
- Fit model parameters (k, p) per phase using scipy.optimize.curve_fit
- At inference: threshold = k / (1 - target_sparsity)^p / seqlen
## Why Choosing the Inverse Power model?
The inverse power model better fits the relationship between sparsity
ratio and threshold_scale_factor.
<img width="2388" height="1082" alt="sparsity_model_analysis"
src="https://github.com/user-attachments/assets/4dfb45d4-8c16-4f15-a878-c8e08a9b6128"
/>
## Runtime Flexibility
- Target sparsity can be changed at inference time without recalibration
- Users can adjust module._sparse_method_instance.target_sparse_ratio
dynamically
- Threshold automatically adapts to sequence length
## Testing
<!-- Mention how have you tested your change if applicable. -->
The calibration results for `Qwen/Qwen3-30B-A3B-Thinking-2507` are shown
below and are mostly consistent with the ground-truth numbers collected
from the kernel side.
```
Prefill Calibration Results:
Model: scale_factor = k / (1 - sparsity)^p
Fitted k: 1003.3990
Fitted p: 1.2589
R-squared: 0.827549
Scale factors for different target sparsities:
Target Scale Factor
---------- ---------------
50% 2401.35
70% 4568.26
80% 7610.98
90% 18214.70
95% 43591.65
```
## Before your PR is "*Ready for review*"
<!-- If you haven't finished some of the above items you can still open
`Draft` PR. -->
- **Make sure you read and follow [Contributor
guidelines](https://github.com/NVIDIA/TensorRT-Model-Optimizer/blob/main/CONTRIBUTING.md)**
and your commits are signed.
- **Is this change backward compatible?**: Yes/No <!--- If No, explain
why. -->
- **Did you write any new necessary tests?**: Yes/No
- **Did you add or update any necessary documentation?**: Yes/No
- **Did you update
[Changelog](https://github.com/NVIDIA/TensorRT-Model-Optimizer/blob/main/CHANGELOG.rst)?**:
Yes/No <!--- Only for new features, API changes, critical bug fixes or
bw breaking changes. -->
## Additional Information
<!-- E.g. related issue. -->
---------
Signed-off-by: Kai Xu <kaix@nvidia.com>
140 lines
4.8 KiB
Python
140 lines
4.8 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""The package setup script for modelopt customizing certain aspects of the installation process."""
|
|
|
|
import setuptools
|
|
from setuptools_scm import get_version
|
|
|
|
# TODO: Set fallback_version to X.Y.Z release version when creating the release branch
|
|
version = get_version(root=".", fallback_version="0.0.0")
|
|
|
|
# Required and optional dependencies ###############################################################
|
|
required_deps = [
|
|
# Common
|
|
"ninja", # for faster building of C++ / CUDA extensions
|
|
"numpy",
|
|
"packaging",
|
|
"pydantic>=2.0",
|
|
"nvidia-ml-py>=12",
|
|
"rich",
|
|
"scipy",
|
|
"tqdm",
|
|
# modelopt.torch
|
|
"pulp",
|
|
"regex",
|
|
"safetensors",
|
|
"torch>=2.6",
|
|
]
|
|
|
|
optional_deps = {
|
|
"onnx": [
|
|
"cppimport",
|
|
"cupy-cuda12x; platform_machine != 'aarch64' and platform_system != 'Darwin'",
|
|
"lief",
|
|
"ml_dtypes", # for bfloat16 conversion
|
|
"onnx-graphsurgeon",
|
|
"onnx~=1.19.0",
|
|
"onnxconverter-common~=1.16.0",
|
|
"onnxruntime~=1.22.0 ; platform_machine == 'aarch64' or platform_system == 'Darwin'",
|
|
"onnxruntime-gpu~=1.22.0 ; platform_machine != 'aarch64' and platform_system != 'Darwin'",
|
|
"onnxscript", # For autocast opset conversion and test_onnx_dynamo_export unit test
|
|
"onnxslim>=0.1.76",
|
|
"polygraphy>=0.49.22",
|
|
],
|
|
"hf": [
|
|
"accelerate>=1.0.0",
|
|
"datasets>=3.0.0",
|
|
"deepspeed>=0.9.6 ; platform_system != 'Darwin' and platform_system != 'Windows'",
|
|
"diffusers>=0.32.2",
|
|
"huggingface_hub>=0.24.0",
|
|
"peft>=0.17.0",
|
|
"transformers>=4.53,<5.0", # Should match modelopt/torch/__init__.py and tox.ini
|
|
"nltk",
|
|
"wonderwords",
|
|
],
|
|
# linter tools
|
|
"dev-lint": [
|
|
"bandit[toml]==1.7.9", # security/compliance checks
|
|
"mypy==1.17.1",
|
|
"pre-commit==4.3.0",
|
|
"ruff==0.12.11",
|
|
],
|
|
# testing
|
|
"dev-test": [
|
|
"coverage",
|
|
"pytest",
|
|
"pytest-cov",
|
|
"pytest-instafail",
|
|
"pytest-timeout",
|
|
"sentencepiece", # For test_unified_export_megatron.py, test_vllm_fakequant_megatron_export.py
|
|
"timm",
|
|
"torchprofile>=0.0.4", # For computing flops of CV models
|
|
"torchvision",
|
|
"torch-geometric",
|
|
"tox>4.18",
|
|
"tox-current-env>=0.0.12",
|
|
],
|
|
# docs
|
|
"dev-docs": [
|
|
"autodoc_pydantic>=2.1.0",
|
|
"sphinx~=8.1.0",
|
|
"sphinx-argparse>=0.5.2",
|
|
"sphinx-autobuild>=2024.10.3",
|
|
"sphinx-copybutton>=0.5.2",
|
|
"sphinx-inline-tabs>=2023.4.21",
|
|
"sphinx-rtd-theme~=3.0.0", # 3.0 does not show version, which we want as Linux & Windows have separate releases
|
|
"sphinx-togglebutton>=0.3.2",
|
|
],
|
|
# build/packaging tools
|
|
"dev-build": [
|
|
"cython",
|
|
"setuptools>=80",
|
|
"setuptools-scm>=8",
|
|
],
|
|
}
|
|
|
|
# create "compound" optional dependencies
|
|
optional_deps["all"] = [
|
|
deps for k in optional_deps if not k.startswith("dev") for deps in optional_deps[k]
|
|
]
|
|
optional_deps["dev"] = [deps for k in optional_deps for deps in optional_deps[k]]
|
|
|
|
|
|
if __name__ == "__main__":
|
|
setuptools.setup(
|
|
name="nvidia-modelopt",
|
|
version=version,
|
|
description="Nvidia Model Optimizer: a unified model optimization and deployment toolkit.",
|
|
long_description="Checkout https://github.com/nvidia/Model-Optimizer for more information.",
|
|
long_description_content_type="text/markdown",
|
|
author="NVIDIA Corporation",
|
|
url="https://github.com/NVIDIA/Model-Optimizer",
|
|
license="Apache 2.0",
|
|
license_files=("LICENSE_HEADER",),
|
|
classifiers=[
|
|
"Programming Language :: Python :: 3",
|
|
"Intended Audience :: Developers",
|
|
"Intended Audience :: Science/Research",
|
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
],
|
|
python_requires=">=3.10,<3.13",
|
|
install_requires=required_deps,
|
|
extras_require=optional_deps,
|
|
packages=setuptools.find_namespace_packages(include=["modelopt*"]),
|
|
package_dir={"": "."},
|
|
package_data={"modelopt": ["**/*.h", "**/*.cpp", "**/*.cu"]},
|
|
)
|