mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
## Summary - add native CUDA packing kernels for IQ1_S and IQ2_XS - load both extensions through the quantization extension module - validate caller metadata and launch bounds before contiguous materialization - normalize non-finite input elements consistently with the Python reference path - accept caller-computed IQ2_XS FP16 superblock scales to avoid a duplicate reduction - share common packing helpers and add direct extension compilation and boundary tests ## PR split This work is split into four focused PRs. Each PR targets `main` and owns a disjoint file set: 1. **Kernel** — [#2448: Add CUDA kernels for IQ packing](https://github.com/NVIDIA/Model-Optimizer/pull/2448) 2. **Quantization** — [#2446: Add IQ quantization codecs and backend](https://github.com/NVIDIA/Model-Optimizer/pull/2446) 3. **Export** — [#2447: Export IQ checkpoints from HF and Megatron](https://github.com/NVIDIA/Model-Optimizer/pull/2447) 4. **Recipes** — [#2449: Add IQ post-training quantization recipes](https://github.com/NVIDIA/Model-Optimizer/pull/2449) The required merge order is #2448, #2446, #2447, then #2449. ## Scope This PR owns only native kernel sources, shared packing helpers, extension loading and build registration, and direct extension tests. It does not contain Python codecs, export code, or recipes. ## GPU test coverage Direct kernel-boundary coverage is included in this PR: - [extension compilation, zero payloads, input validation, and row-alignment checks](https://github.com/NVIDIA/Model-Optimizer/blob/74e94db9601870e3569c7c8e73506a2f08c29da8/tests/gpu/_extensions/test_torch_extensions.py) Pack/dequantize numerical, native/reference byte-parity, and non-finite-policy tests are owned by the quantization PR: [IQ1_S](https://github.com/NVIDIA/Model-Optimizer/blob/e8d937081d8cd01cf8e44d43915df443b79deb17/tests/gpu/torch/quantization/test_iq1_s_cuda.py) and [IQ2_XS](https://github.com/NVIDIA/Model-Optimizer/blob/e8d937081d8cd01cf8e44d43915df443b79deb17/tests/gpu/torch/quantization/test_iq2_xs_cuda.py). ## Dependency behavior On `main`, this PR provides optional CUDA extension loaders and direct extension tests. The Python encoders and fallback dispatch land in #2446. Until #2446 lands, no quantization path calls these getters, so a load failure reports only that the extension is unavailable. The IQ2_XS packer accepts one caller-computed FP16 scale per 256-value block. #2446 owns that predictor and passes the same values to the native and reference encoders. ## Provenance - The CUDA kernels were independently written. - They implement the packed-format contract and sign-parity convention from the pinned [llama.cpp definition](https://github.com/ggml-org/llama.cpp/blob/9b05354ec6fb58b4e665e9a39ebc40285c015638/ggml/src/ggml-common.h). - `16.875 = 15 × (1 + 1/8)` is a derived IQ1_S constant. - `0.125` is part of the encoded format. - `0.61` is our empirical IQ1_S scale predictor, not copied from upstream code. - The IQ2_XS predictor constants are owned by #2446 and are not duplicated in this kernel. Human review is still required to confirm that the attribution and license treatment are sufficient. ## Validation - repository hooks, including native formatting, pass for all changed files - extension loader and test modules compile as Python - all 20 direct-extension and CUDA integration test cases collect locally - CUDA runtime execution is delegated to GPU CI because the local host is macOS - restricted-term scan passes --------- Signed-off-by: Hung-Yueh Chiang <hungyuehc@nvidia.com> Signed-off-by: Chenjie Luo <chenjiel@nvidia.com> Co-authored-by: Chenjie Luo <chenjiel@nvidia.com> Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
385 lines
14 KiB
TOML
385 lines
14 KiB
TOML
####################################################################################################
|
|
############################### BUILD CONFIGURATION ##############################################
|
|
####################################################################################################
|
|
[build-system]
|
|
requires = ["setuptools>=80", "setuptools-scm>=8,<10"]
|
|
build-backend = "setuptools.build_meta"
|
|
|
|
[tool.setuptools_scm]
|
|
fallback_version = "0.0.0"
|
|
|
|
|
|
####################################################################################################
|
|
############################### PROJECT METADATA #################################################
|
|
####################################################################################################
|
|
[project]
|
|
name = "nvidia-modelopt"
|
|
dynamic = ["version"]
|
|
description = "Nvidia Model Optimizer: A unified library of SOTA model optimization techniques like quantization, pruning, Neural Architecture Search (NAS), distillation, speculative decoding, etc. It compresses deep learning models for downstream deployment frameworks like TensorRT-LLM, TensorRT, vLLM, etc. to optimize inference speed."
|
|
readme = { text = "Checkout https://github.com/nvidia/Model-Optimizer for more information.", content-type = "text/markdown" }
|
|
license = "Apache-2.0"
|
|
license-files = ["LICENSE_HEADER"]
|
|
requires-python = ">=3.10,<3.15"
|
|
authors = [{ name = "NVIDIA Corporation" }]
|
|
classifiers = [
|
|
"Programming Language :: Python :: 3",
|
|
"Programming Language :: Python :: 3.10",
|
|
"Programming Language :: Python :: 3.11",
|
|
"Programming Language :: Python :: 3.12",
|
|
"Programming Language :: Python :: 3.13",
|
|
"Intended Audience :: Developers",
|
|
"Intended Audience :: Science/Research",
|
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
]
|
|
dependencies = [
|
|
# Common
|
|
"ninja", # for faster building of C++ / CUDA extensions
|
|
"numpy",
|
|
"nvidia-ml-py>=12",
|
|
"packaging",
|
|
"setuptools>=80", # torch.utils.cpp_extension imports setuptools at load time
|
|
"torch>=2.8",
|
|
"tqdm",
|
|
# modelopt.torch
|
|
"PyYAML>=6.0",
|
|
"omegaconf>=2.3.0",
|
|
"pulp<4.0", # breaking changes in upcoming 4.0 release
|
|
"pydantic>=2.0",
|
|
"regex",
|
|
"rich",
|
|
"safetensors",
|
|
"scipy",
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
onnx = [
|
|
"cppimport",
|
|
"cupy-cuda12x; platform_machine != 'aarch64' and platform_machine != 'ARM64' and platform_system != 'Darwin'",
|
|
"lief",
|
|
"ml_dtypes",
|
|
"onnx-graphsurgeon>=0.6.1",
|
|
"onnx~=1.21.0",
|
|
"onnxconverter-common~=1.16.0",
|
|
# Python 3.10 support for ModelOpt ONNX is deprecated and will be removed in a future release.
|
|
# Native Windows ARM64 does not support ModelOpt ONNX on Python 3.10.
|
|
# ORT for Windows x64
|
|
"onnxruntime-gpu==1.22.0; platform_system == 'Windows' and platform_machine == 'AMD64'",
|
|
# ORT host and standalone TensorRT-RTX ABI EP for native Windows ARM64.
|
|
"onnxruntime~=1.24.2; python_version > '3.10' and platform_system == 'Windows' and platform_machine == 'ARM64'",
|
|
"onnxruntime-ep-nv-tensorrt-rtx-cu13==0.4.0; python_version > '3.10' and platform_system == 'Windows' and platform_machine == 'ARM64'",
|
|
# ORT with Python <= 3.10 on supported platforms
|
|
"onnxruntime~=1.22.0; python_version <= '3.10' and (platform_machine == 'aarch64' or platform_system == 'Darwin')",
|
|
"onnxruntime-gpu~=1.22.0; python_version <= '3.10' and platform_machine != 'aarch64' and platform_system != 'Darwin' and platform_system != 'Windows'",
|
|
# ORT with Python > 3.10
|
|
"onnxruntime~=1.24.2; python_version > '3.10' and (platform_machine == 'aarch64' or platform_system == 'Darwin')",
|
|
"onnxruntime-gpu~=1.24.2; python_version > '3.10' and platform_machine != 'aarch64' and platform_system != 'Darwin' and platform_system != 'Windows'",
|
|
"onnxscript",
|
|
"onnxslim>=0.1.76",
|
|
"polygraphy>=0.53.4",
|
|
]
|
|
hf = [
|
|
"accelerate>=1.0.0",
|
|
"datasets>=3.0.0",
|
|
"deepspeed>=0.9.6; platform_system != 'Darwin' and platform_system != 'Windows'",
|
|
"diffusers>=0.32.2",
|
|
"huggingface_hub>=0.24.0",
|
|
"nltk",
|
|
"peft>=0.17.0",
|
|
"sentencepiece>=0.2.1", # Also implicitly used in test_unified_export_megatron, test_vllm_fakequant_megatron_export
|
|
"tiktoken",
|
|
"transformers>=4.57,<5.15", # Should match modelopt/torch/__init__.py and noxfile.py
|
|
"wonderwords",
|
|
]
|
|
|
|
puzzletron = [ # Dependedencies for modelopt.torch.puzzletron subpackage
|
|
"fire",
|
|
"hydra-core~=1.3.4",
|
|
"immutabledict",
|
|
"lru-dict",
|
|
"pandas",
|
|
"typeguard",
|
|
]
|
|
dev-lint = [
|
|
"bandit[toml]==1.7.9", # security/compliance checks
|
|
"mypy==2.1.0",
|
|
"pre-commit==4.6.0",
|
|
"ruff==0.15.20",
|
|
]
|
|
# Docs are only built on Python >=3.12 (sphinx 9.x requires it); markers keep `uv lock` from resolving these for 3.10/3.11, which we don't care about for building docs.
|
|
dev-docs = [
|
|
"autodoc_pydantic~=2.2.0; python_version >= '3.12'",
|
|
"sphinx~=9.1.0; python_version >= '3.12'",
|
|
"sphinx-argparse~=0.6.0; python_version >= '3.12'",
|
|
"sphinx-autobuild==2025.8.25; python_version >= '3.12'",
|
|
"sphinx-copybutton~=0.5.2; python_version >= '3.12'",
|
|
"sphinx-inline-tabs==2025.12.21.14; python_version >= '3.12'",
|
|
"shibuya~=2026.7.0; python_version >= '3.12'",
|
|
"sphinx-togglebutton~=0.4.0; python_version >= '3.12'",
|
|
]
|
|
dev-test = [
|
|
"coverage[toml]~=7.14.0", # a1_coverage.pth for subprocess tracking requires this
|
|
"nox",
|
|
"pytest~=9.1.0",
|
|
"pytest-cov~=7.1.0",
|
|
"pytest-instafail==0.5.0",
|
|
"pytest-timeout~=2.4.0",
|
|
"tomli; python_version < '3.11'",
|
|
"uv",
|
|
# test-specific dependencies
|
|
"psutil", # tools/resource_monitor.py sidecar (tests/unit/tools)
|
|
"timm",
|
|
"torchprofile==0.0.4", # optional dependency for modelopt.torch
|
|
"torchvision",
|
|
"torch-geometric",
|
|
]
|
|
# Compound extras via self-references
|
|
mlflow = [ # Dependencies for modelopt.torch.utils.mlflow run tracking
|
|
"mlflow-skinny>=2.9", # client only; the tracking server's own deps are not needed here
|
|
]
|
|
|
|
# mlflow is deliberately not in `all`: it is opt-in tracking for the examples, and the
|
|
# full client would land in the documented `nvidia-modelopt[all]` install path.
|
|
all = ["nvidia-modelopt[onnx,hf,puzzletron]"]
|
|
dev = ["nvidia-modelopt[all,mlflow,dev-docs,dev-lint,dev-test]"]
|
|
|
|
[project.urls]
|
|
Homepage = "https://github.com/NVIDIA/Model-Optimizer"
|
|
|
|
[tool.setuptools.packages.find]
|
|
include = ["modelopt*"]
|
|
|
|
[tool.setuptools.package-data]
|
|
modelopt = ["**/*.cpp", "**/*.cu", "**/*.cuh", "**/*.h"]
|
|
modelopt_recipes = ["**/*.yml", "**/*.yaml"]
|
|
|
|
[tool.setuptools.exclude-package-data]
|
|
# The recipe library keeps two backward-compat symlinks: the top-level
|
|
# ``huggingface`` -> ``model_type`` rename alias, and the nested
|
|
# ``model_type/models`` -> ``../models`` alias. The recursive package-data glob above
|
|
# follows both, so drop the aliased copies here to avoid shipping every recipe twice;
|
|
# setuptools' exclude glob is non-recursive, hence the explicit per-depth entries.
|
|
# MANIFEST.in prunes the symlink dir entries themselves (build_py can't copy a
|
|
# symlink-to-dir). Old huggingface/... --recipe paths keep working via the loader
|
|
# alias in modelopt/recipe/loader.py.
|
|
modelopt_recipes = [
|
|
"huggingface",
|
|
"model_type/models",
|
|
# Architecture recipes duplicated via ``huggingface`` -> ``model_type``.
|
|
"huggingface/*/*/*.yaml", "huggingface/*/*/*.yml",
|
|
# Checkpoint mirrors duplicated via ``huggingface/models`` and
|
|
# ``model_type/models`` (both resolve to the real top-level ``models/`` tier).
|
|
"huggingface/models/*/*/*/*.yaml", "huggingface/models/*/*/*/*.yml",
|
|
"huggingface/models/*/*/*/*/*.yaml", "huggingface/models/*/*/*/*/*.yml",
|
|
"model_type/models/*/*/*/*.yaml", "model_type/models/*/*/*/*.yml",
|
|
"model_type/models/*/*/*/*/*.yaml", "model_type/models/*/*/*/*/*.yml",
|
|
]
|
|
|
|
[tool.uv]
|
|
managed = true
|
|
# override-dependencies = ["torch; sys_platform == 'never'"]
|
|
|
|
|
|
####################################################################################################
|
|
############################### LINTING, FORMATTING CONFIGURATION ################################
|
|
####################################################################################################
|
|
[tool.ruff]
|
|
target-version = "py310"
|
|
line-length = 100 # Line length limit for code
|
|
fix = true
|
|
|
|
[tool.ruff.format]
|
|
# Like Black, respect magic trailing commas.
|
|
skip-magic-trailing-comma = false
|
|
docstring-code-format = true
|
|
# Set the line length limit used when formatting code snippets in docstrings.
|
|
docstring-code-line-length = "dynamic"
|
|
|
|
[tool.ruff.lint]
|
|
# See available rules at https://docs.astral.sh/ruff/rules/
|
|
# Flake8 is equivalent to pycodestyle + pyflakes + mccabe.
|
|
select = [
|
|
"C4", # Flake8 comprehensions
|
|
"D", # pydocstyle
|
|
"E", # pycodestyle errors
|
|
"F", # pyflakes
|
|
"FURB", # refurb
|
|
"I", # isort
|
|
"ISC", # flake8-implicit-str-concat
|
|
"N", # pep8 naming
|
|
"PERF", # Perflint
|
|
"PGH", # pygrep-hooks
|
|
"PIE", # flake8-pie
|
|
"PLE", # pylint errors
|
|
"PLR", # pylint refactor
|
|
"PT", # flake8-pytest-style
|
|
"RUF", # ruff
|
|
"SIM", # flake8-simplify
|
|
"TC", # flake8-type-checking
|
|
"UP", # pyupgrade
|
|
"W", # pycodestyle warnings
|
|
]
|
|
extend-ignore = [
|
|
"D105",
|
|
"D417",
|
|
"N812",
|
|
"PLR0402",
|
|
"PLR0912",
|
|
"PLR0913",
|
|
"PLR0915",
|
|
"PLR2004",
|
|
"PLR0911",
|
|
"PT011",
|
|
"PT018",
|
|
"PT028",
|
|
"PT030",
|
|
"RUF002",
|
|
"RUF012",
|
|
"RUF059",
|
|
"SIM102",
|
|
"SIM108",
|
|
"SIM115",
|
|
]
|
|
|
|
|
|
[tool.ruff.lint.per-file-ignores]
|
|
"__init__.py" = ["F401", "F403"]
|
|
"examples/*" = ["D"]
|
|
"noxfile.py" = ["D", "E501"]
|
|
"tests/*" = ["B017", "D", "E402", "PT012"]
|
|
"plugins/modelopt/skills/*/tests/test_*.py" = ["D", "E402"] # Skill test scripts: docstring (D) + sys.path import-order (E402) exemptions
|
|
"*/_[a-zA-Z]*" = ["D"] # Private packages (_abc/*.py) or modules (_xyz.py)
|
|
"*.ipynb" = ["D", "E501"] # Ignore missing docstrings or line length for Jupyter notebooks
|
|
"modelopt/torch/kernels/*" = ["N803", "N806", "E731"] # triton style
|
|
"modelopt/torch/puzzletron/*" = [
|
|
"C4",
|
|
"D",
|
|
"E",
|
|
"F",
|
|
"N",
|
|
"PERF",
|
|
"RUF",
|
|
"SIM",
|
|
"UP",
|
|
] # TODO: Disabled for now, will enable once newer puzzletron code with ruff fixes is migrated
|
|
|
|
[tool.ruff.lint.pycodestyle]
|
|
max-line-length = 120 # Line length limit for comments and docstrings
|
|
|
|
|
|
[tool.ruff.lint.pydocstyle]
|
|
convention = "google"
|
|
|
|
|
|
[tool.ruff.lint.isort]
|
|
known-first-party = ["modelopt"]
|
|
known-third-party = ["vllm"]
|
|
split-on-trailing-comma = false
|
|
|
|
|
|
[tool.mypy]
|
|
files = "."
|
|
python_version = "3.10"
|
|
install_types = true
|
|
non_interactive = true
|
|
show_error_codes = true
|
|
disable_error_code = [
|
|
"assignment",
|
|
"operator",
|
|
"has-type",
|
|
"var-annotated",
|
|
"override",
|
|
]
|
|
enable_error_code = [
|
|
"ignore-without-code", # require `# type: ignore[code]` instead of bare `# type: ignore`
|
|
"deprecated", # use of `@deprecated`-marked APIs
|
|
# TODO: enable in a follow-up (each surfaces many errors needing case-by-case fixes):
|
|
# "possibly-undefined", "redundant-expr", "truthy-bool"
|
|
]
|
|
explicit_package_bases = true
|
|
namespace_packages = true
|
|
# strict checks
|
|
strict = true
|
|
disallow_subclassing_any = false
|
|
disallow_untyped_decorators = false
|
|
disallow_any_generics = false
|
|
disallow_untyped_calls = false
|
|
disallow_incomplete_defs = false
|
|
disallow_untyped_defs = false
|
|
warn_return_any = false
|
|
|
|
|
|
[[tool.mypy.overrides]]
|
|
module = ["tests.*"]
|
|
ignore_errors = true
|
|
|
|
[[tool.mypy.overrides]]
|
|
module = ["examples.*"]
|
|
disable_error_code = ["attr-defined"]
|
|
|
|
# Vendored verbatim from NVIDIA-NeMo/Automodel (Apache-2.0); kept faithful to upstream rather
|
|
# than annotated to modelopt's strict mypy (e.g. worker-global ``X | None`` initialized lazily).
|
|
[[tool.mypy.overrides]]
|
|
module = ["examples.diffusers.fastgen.preprocess.*"]
|
|
ignore_errors = true
|
|
|
|
[tool.bandit]
|
|
exclude_dirs = [".github/", "examples/", "noxfile.py", "tests/"]
|
|
# Do not change `skips`. It should be consistent with NVIDIA's Wheel-CI-CD bandit.yml config.
|
|
# Use of `# nosec BXXX` requires special approval
|
|
skips = [
|
|
"B101", # assert_used
|
|
"B110", # try_except_pass
|
|
"B112", # try_except_continue
|
|
"B303", # MD2, MD4, MD5, or SHA1
|
|
"B311", # random
|
|
]
|
|
|
|
|
|
####################################################################################################
|
|
############################### TESTING CONFIGURATION ############################################
|
|
####################################################################################################
|
|
[tool.pytest.ini_options]
|
|
# Default additional options
|
|
# Show a short test summary info for all except passed tests with -ra flag
|
|
# print execution time for 50 slowest tests and generate coverage reports
|
|
addopts = "-v -ra --instafail --cov-report=term-missing --cov-report=html --cov-report=xml:coverage.xml --cov-config=pyproject.toml --durations=50 --strict-markers"
|
|
pythonpath = ["tests/"]
|
|
# Apply per-test timeouts (see tests/conftest.py) to the test call only, not fixture setup/teardown
|
|
timeout_func_only = true
|
|
markers = [
|
|
"integration: Tests that require external services or other non-hermetic dependencies",
|
|
"manual: Only run when --run-manual is given",
|
|
"release: Regression tests that should be run before every release",
|
|
]
|
|
|
|
|
|
[tool.coverage.run]
|
|
branch = false
|
|
parallel = true
|
|
concurrency = ["multiprocessing", "thread"]
|
|
source_pkgs = ["modelopt", "modelopt_recipes"]
|
|
|
|
|
|
[tool.coverage.report]
|
|
skip_covered = true
|
|
ignore_errors = true
|
|
exclude_lines = [
|
|
"pragma: no cover",
|
|
# Don't complain about missing debug or verbose code
|
|
"def __repr__",
|
|
"if verbose",
|
|
# Don't complain if tests don't hit defensive exception handling code
|
|
"raise AssertionError",
|
|
"raise NotImplementedError",
|
|
"raise RuntimeError",
|
|
"raise ValueError",
|
|
"raise KeyError",
|
|
"raise AttributeError",
|
|
"except ImportError",
|
|
# Don't complain if non-runnable code isn't run
|
|
"if __name__ == \"__main__\":",
|
|
"if TYPE_CHECKING:",
|
|
# Don't complain about abstract methods, they aren't run
|
|
"@(abc\\.)?abstractmethod",
|
|
]
|