Files
Model-Optimizer/pyproject.toml
T
Chad Voegele 28dc117594 Add Day 0 Sub-agent Roles (#2006)
### What does this PR do?

Type of change: new feature

Adding sub-agent definition for Day 0 workflows.

I'm adding subagents as an alternative, while we evaluate which is
better. They are duplicated because Claude & Codex have different
formats & different locations. Deriving from a shared vendor-neutral
format at build-time is too complicated.

### Usage

Tell your agent "Use the modelopt_model_quantizer agent to quantize the
model"

### Testing

Ran a trial using Qwen-2.5, Muse Glimmer, Qwen-2.8, and GLM-5.3-Flash.

### Before your PR is "*Ready for review*"

Make sure you read and follow [Contributor
guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md)
and your commits are signed (`git commit -s -S`).

Make sure you read and follow the [Security Best
Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors)
(e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(...,
weights_only=False)`, `pickle`, etc.).

- Is this change backward compatible?: ✅
- If you copied code from any other sources or added a new PIP
dependency, did you follow guidance in `CONTRIBUTING.md`: ✅
- Did you write any new necessary tests?: ✅
- Did you update
[Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?:
✅ / ❌ / N/A <!--- Only for new features, API changes, critical bug fixes
or backward incompatible changes. -->
- Did you get Claude approval on this PR?: ✅ / ❌ / N/A <!--- Run
`/claude review`. NVIDIA org members can self-trigger for complex
changes; orthogonal to CodeRabbit. -->

### Additional Information
See design doc.


<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

## New Features
- Added specialized Model Optimizer agents for downloading,
quantization, recipe search, evaluation, deployment, and performance
benchmarking.
- Added coordinated Codex and Claude Code access with validation
requirements, artifact preservation, and concise handoffs.
- Improved skill discovery across plugin and repository installations.

## Documentation
- Clarified agent discovery, layout, and canonical editing locations.
- Updated configuration guidance for supported agent definitions.

## Tests
- Added synchronization checks for agent definitions and links.
- Improved test compatibility for Python versions below 3.11.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->

---------

Signed-off-by: Chad Voegele <cvoegele@nvidia.com>
2026-09-10 17:30:37 +00:00

376 lines
13 KiB
TOML

####################################################################################################
############################### BUILD CONFIGURATION ##############################################
####################################################################################################
[build-system]
requires = ["setuptools>=80", "setuptools-scm>=8,<10"]
build-backend = "setuptools.build_meta"
[tool.setuptools_scm]
fallback_version = "0.0.0"
####################################################################################################
############################### PROJECT METADATA #################################################
####################################################################################################
[project]
name = "nvidia-modelopt"
dynamic = ["version"]
description = "Nvidia Model Optimizer: A unified library of SOTA model optimization techniques like quantization, pruning, Neural Architecture Search (NAS), distillation, speculative decoding, etc. It compresses deep learning models for downstream deployment frameworks like TensorRT-LLM, TensorRT, vLLM, etc. to optimize inference speed."
readme = { text = "Checkout https://github.com/nvidia/Model-Optimizer for more information.", content-type = "text/markdown" }
license = "Apache-2.0"
license-files = ["LICENSE_HEADER"]
requires-python = ">=3.10,<3.15"
authors = [{ name = "NVIDIA Corporation" }]
classifiers = [
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.10",
"Programming Language :: Python :: 3.11",
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.13",
"Intended Audience :: Developers",
"Intended Audience :: Science/Research",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
]
dependencies = [
# Common
"ninja", # for faster building of C++ / CUDA extensions
"numpy",
"nvidia-ml-py>=12",
"packaging",
"setuptools>=80", # torch.utils.cpp_extension imports setuptools at load time
"torch>=2.8",
"tqdm",
# modelopt.torch
"PyYAML>=6.0",
"omegaconf>=2.3.0",
"pulp<4.0", # breaking changes in upcoming 4.0 release
"pydantic>=2.0",
"regex",
"rich",
"safetensors",
"scipy",
]
[project.optional-dependencies]
onnx = [
"cppimport",
"cupy-cuda12x; platform_machine != 'aarch64' and platform_machine != 'ARM64' and platform_system != 'Darwin'",
"lief",
"ml_dtypes",
"onnx-graphsurgeon>=0.6.1",
"onnx~=1.21.0",
"onnxconverter-common~=1.16.0",
# Python 3.10 support for ModelOpt ONNX is deprecated and will be removed in a future release.
# Native Windows ARM64 does not support ModelOpt ONNX on Python 3.10.
# ORT for Windows x64
"onnxruntime-gpu==1.22.0; platform_system == 'Windows' and platform_machine == 'AMD64'",
# ORT host and standalone TensorRT-RTX ABI EP for native Windows ARM64.
"onnxruntime~=1.24.2; python_version > '3.10' and platform_system == 'Windows' and platform_machine == 'ARM64'",
"onnxruntime-ep-nv-tensorrt-rtx-cu13==0.4.0; python_version > '3.10' and platform_system == 'Windows' and platform_machine == 'ARM64'",
# ORT with Python <= 3.10 on supported platforms
"onnxruntime~=1.22.0; python_version <= '3.10' and (platform_machine == 'aarch64' or platform_system == 'Darwin')",
"onnxruntime-gpu~=1.22.0; python_version <= '3.10' and platform_machine != 'aarch64' and platform_system != 'Darwin' and platform_system != 'Windows'",
# ORT with Python > 3.10
"onnxruntime~=1.24.2; python_version > '3.10' and (platform_machine == 'aarch64' or platform_system == 'Darwin')",
"onnxruntime-gpu~=1.24.2; python_version > '3.10' and platform_machine != 'aarch64' and platform_system != 'Darwin' and platform_system != 'Windows'",
"onnxscript",
"onnxslim>=0.1.76",
"polygraphy>=0.53.4",
]
hf = [
"accelerate>=1.0.0",
"datasets>=3.0.0",
"deepspeed>=0.9.6; platform_system != 'Darwin' and platform_system != 'Windows'",
"diffusers>=0.32.2",
"huggingface_hub>=0.24.0",
"nltk",
"peft>=0.17.0",
"sentencepiece>=0.2.1", # Also implicitly used in test_unified_export_megatron, test_vllm_fakequant_megatron_export
"tiktoken",
"transformers>=4.57,<5.15", # Should match modelopt/torch/__init__.py and noxfile.py
"wonderwords",
]
puzzletron = [ # Dependedencies for modelopt.torch.puzzletron subpackage
"fire",
"hydra-core~=1.3.4",
"immutabledict",
"lru-dict",
"pandas",
"typeguard",
]
dev-lint = [
"bandit[toml]==1.7.9", # security/compliance checks
"mypy==2.1.0",
"pre-commit==4.6.0",
"ruff==0.15.20",
]
# Docs are only built on Python >=3.12 (sphinx 9.x requires it); markers keep `uv lock` from resolving these for 3.10/3.11, which we don't care about for building docs.
dev-docs = [
"autodoc_pydantic~=2.2.0; python_version >= '3.12'",
"sphinx~=9.1.0; python_version >= '3.12'",
"sphinx-argparse~=0.6.0; python_version >= '3.12'",
"sphinx-autobuild==2025.8.25; python_version >= '3.12'",
"sphinx-copybutton~=0.5.2; python_version >= '3.12'",
"sphinx-inline-tabs==2025.12.21.14; python_version >= '3.12'",
"shibuya~=2026.7.0; python_version >= '3.12'",
"sphinx-togglebutton~=0.4.0; python_version >= '3.12'",
]
dev-test = [
"coverage[toml]~=7.14.0", # a1_coverage.pth for subprocess tracking requires this
"nox",
"pytest~=9.1.0",
"pytest-cov~=7.1.0",
"pytest-instafail==0.5.0",
"pytest-timeout~=2.4.0",
"tomli; python_version < '3.11'",
"uv",
# test-specific dependencies
"psutil", # tools/resource_monitor.py sidecar (tests/unit/tools)
"timm",
"torchprofile==0.0.4", # optional dependency for modelopt.torch
"torchvision",
"torch-geometric",
]
# Compound extras via self-references
mlflow = [ # Dependencies for modelopt.torch.utils.mlflow run tracking
"mlflow-skinny>=2.9", # client only; the tracking server's own deps are not needed here
]
# mlflow is deliberately not in `all`: it is opt-in tracking for the examples, and the
# full client would land in the documented `nvidia-modelopt[all]` install path.
all = ["nvidia-modelopt[onnx,hf,puzzletron]"]
dev = ["nvidia-modelopt[all,mlflow,dev-docs,dev-lint,dev-test]"]
[project.urls]
Homepage = "https://github.com/NVIDIA/Model-Optimizer"
[tool.setuptools.packages.find]
include = ["modelopt*"]
[tool.setuptools.package-data]
modelopt = ["**/*.h", "**/*.cpp", "**/*.cu"]
modelopt_recipes = ["**/*.yml", "**/*.yaml"]
[tool.setuptools.exclude-package-data]
# huggingface/models is a backward-compat symlink to ../models. The recursive
# package-data glob above follows it, so drop the aliased copies here to avoid
# shipping every checkpoint recipe twice; setuptools' exclude glob is non-recursive,
# hence the explicit org/model/task depths. MANIFEST.in prunes the symlink dir entry
# itself (build_py can't copy a symlink-to-dir). Old huggingface/models/... --recipe
# paths keep working via the loader alias in modelopt/recipe/loader.py.
modelopt_recipes = [
"huggingface/models",
"huggingface/models/*/*/*/*.yaml", "huggingface/models/*/*/*/*.yml",
"huggingface/models/*/*/*/*/*.yaml", "huggingface/models/*/*/*/*/*.yml",
]
[tool.uv]
managed = true
# override-dependencies = ["torch; sys_platform == 'never'"]
####################################################################################################
############################### LINTING, FORMATTING CONFIGURATION ################################
####################################################################################################
[tool.ruff]
target-version = "py310"
line-length = 100 # Line length limit for code
fix = true
[tool.ruff.format]
# Like Black, respect magic trailing commas.
skip-magic-trailing-comma = false
docstring-code-format = true
# Set the line length limit used when formatting code snippets in docstrings.
docstring-code-line-length = "dynamic"
[tool.ruff.lint]
# See available rules at https://docs.astral.sh/ruff/rules/
# Flake8 is equivalent to pycodestyle + pyflakes + mccabe.
select = [
"C4", # Flake8 comprehensions
"D", # pydocstyle
"E", # pycodestyle errors
"F", # pyflakes
"FURB", # refurb
"I", # isort
"ISC", # flake8-implicit-str-concat
"N", # pep8 naming
"PERF", # Perflint
"PGH", # pygrep-hooks
"PIE", # flake8-pie
"PLE", # pylint errors
"PLR", # pylint refactor
"PT", # flake8-pytest-style
"RUF", # ruff
"SIM", # flake8-simplify
"TC", # flake8-type-checking
"UP", # pyupgrade
"W", # pycodestyle warnings
]
extend-ignore = [
"D105",
"D417",
"N812",
"PLR0402",
"PLR0912",
"PLR0913",
"PLR0915",
"PLR2004",
"PLR0911",
"PT011",
"PT018",
"PT028",
"PT030",
"RUF002",
"RUF012",
"RUF059",
"SIM102",
"SIM108",
"SIM115",
]
[tool.ruff.lint.per-file-ignores]
"__init__.py" = ["F401", "F403"]
"examples/*" = ["D"]
"noxfile.py" = ["D", "E501"]
"tests/*" = ["B017", "D", "E402", "PT012"]
"plugins/modelopt/skills/*/tests/test_*.py" = ["D", "E402"] # Skill test scripts: docstring (D) + sys.path import-order (E402) exemptions
"*/_[a-zA-Z]*" = ["D"] # Private packages (_abc/*.py) or modules (_xyz.py)
"*.ipynb" = ["D", "E501"] # Ignore missing docstrings or line length for Jupyter notebooks
"modelopt/torch/kernels/*" = ["N803", "N806", "E731"] # triton style
"modelopt/torch/puzzletron/*" = [
"C4",
"D",
"E",
"F",
"N",
"PERF",
"RUF",
"SIM",
"UP",
] # TODO: Disabled for now, will enable once newer puzzletron code with ruff fixes is migrated
[tool.ruff.lint.pycodestyle]
max-line-length = 120 # Line length limit for comments and docstrings
[tool.ruff.lint.pydocstyle]
convention = "google"
[tool.ruff.lint.isort]
known-first-party = ["modelopt"]
known-third-party = ["vllm"]
split-on-trailing-comma = false
[tool.mypy]
files = "."
python_version = "3.10"
install_types = true
non_interactive = true
show_error_codes = true
disable_error_code = [
"assignment",
"operator",
"has-type",
"var-annotated",
"override",
]
enable_error_code = [
"ignore-without-code", # require `# type: ignore[code]` instead of bare `# type: ignore`
"deprecated", # use of `@deprecated`-marked APIs
# TODO: enable in a follow-up (each surfaces many errors needing case-by-case fixes):
# "possibly-undefined", "redundant-expr", "truthy-bool"
]
explicit_package_bases = true
namespace_packages = true
# strict checks
strict = true
disallow_subclassing_any = false
disallow_untyped_decorators = false
disallow_any_generics = false
disallow_untyped_calls = false
disallow_incomplete_defs = false
disallow_untyped_defs = false
warn_return_any = false
[[tool.mypy.overrides]]
module = ["tests.*"]
ignore_errors = true
[[tool.mypy.overrides]]
module = ["examples.*"]
disable_error_code = ["attr-defined"]
# Vendored verbatim from NVIDIA-NeMo/Automodel (Apache-2.0); kept faithful to upstream rather
# than annotated to modelopt's strict mypy (e.g. worker-global ``X | None`` initialized lazily).
[[tool.mypy.overrides]]
module = ["examples.diffusers.fastgen.preprocess.*"]
ignore_errors = true
[tool.bandit]
exclude_dirs = [".github/", "examples/", "noxfile.py", "tests/"]
# Do not change `skips`. It should be consistent with NVIDIA's Wheel-CI-CD bandit.yml config.
# Use of `# nosec BXXX` requires special approval
skips = [
"B101", # assert_used
"B110", # try_except_pass
"B112", # try_except_continue
"B303", # MD2, MD4, MD5, or SHA1
"B311", # random
]
####################################################################################################
############################### TESTING CONFIGURATION ############################################
####################################################################################################
[tool.pytest.ini_options]
# Default additional options
# Show a short test summary info for all except passed tests with -ra flag
# print execution time for 50 slowest tests and generate coverage reports
addopts = "-v -ra --instafail --cov-report=term-missing --cov-report=html --cov-report=xml:coverage.xml --cov-config=pyproject.toml --durations=50 --strict-markers"
pythonpath = ["tests/"]
# Apply per-test timeouts (see tests/conftest.py) to the test call only, not fixture setup/teardown
timeout_func_only = true
markers = [
"integration: Tests that require external services or other non-hermetic dependencies",
"manual: Only run when --run-manual is given",
"release: Regression tests that should be run before every release",
]
[tool.coverage.run]
branch = false
parallel = true
concurrency = ["multiprocessing", "thread"]
source_pkgs = ["modelopt", "modelopt_recipes"]
[tool.coverage.report]
skip_covered = true
ignore_errors = true
exclude_lines = [
"pragma: no cover",
# Don't complain about missing debug or verbose code
"def __repr__",
"if verbose",
# Don't complain if tests don't hit defensive exception handling code
"raise AssertionError",
"raise NotImplementedError",
"raise RuntimeError",
"raise ValueError",
"raise KeyError",
"raise AttributeError",
"except ImportError",
# Don't complain if non-runnable code isn't run
"if __name__ == \"__main__\":",
"if TYPE_CHECKING:",
# Don't complain about abstract methods, they aren't run
"@(abc\\.)?abstractmethod",
]