mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
## What does this PR do? - Upgrade CICD test containers to latest - Enable torch 2.10 testing in CICD ## Testing <!-- Mention how have you tested your change if applicable. --> CI/CD in this PR should pass <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added support for mixed-precision gradient handling with FSDP2. * **Documentation** * Updated Linux installation guide with CUDA 13.x support and cupy dependency guidance. * **Chores** * Updated CI/CD workflows and test infrastructure to support PyTorch 2.10 and CUDA 13. * Updated container image versions and test environment configurations. * Updated TensorRT-LLM version requirements. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com>
136 lines
4.0 KiB
INI
136 lines
4.0 KiB
INI
[tox]
|
|
envlist=
|
|
pre-commit-all
|
|
py312-torch210-tf_latest-unit
|
|
cuda13-gpu
|
|
cuda13-gpu-megatron
|
|
skipsdist = True
|
|
toxworkdir = /tmp/{env:USER}-modelopt-tox
|
|
|
|
|
|
############################
|
|
# CPU Unit test environments
|
|
############################
|
|
[testenv:{py310,py311,py312}-torch{26,27,28,29,210}-tf_{min,latest}-unit]
|
|
deps =
|
|
# torch version auto-selected based on torchvision version
|
|
torch26: torchvision~=0.21.0
|
|
torch27: torchvision~=0.22.0
|
|
torch28: torchvision~=0.23.0
|
|
torch29: torchvision~=0.24.0
|
|
torch210: torchvision~=0.25.0
|
|
|
|
# Install megatron-core for special unit tests
|
|
megatron-core
|
|
|
|
-e .[all,dev-test]
|
|
|
|
# Should match setup.py
|
|
tf_min: transformers~=4.53.0
|
|
commands =
|
|
python -m pytest tests/unit {env:COV_ARGS:}
|
|
|
|
|
|
#####################################################################
|
|
# Environment to run unit tests with subset of dependencies installed
|
|
#####################################################################
|
|
[testenv:{py310,py311,py312}-partial-unit-{onnx,torch,torch_deploy}]
|
|
allowlist_externals =
|
|
bash, rm
|
|
deps =
|
|
# Make sure torch 2.10 is used
|
|
torchvision~=0.25.0
|
|
|
|
# ONNX unit tests heavily rely on torch / torchvision
|
|
onnx: .[onnx,dev-test]
|
|
onnx: torchvision
|
|
|
|
# Install megatron-core to test torch-only install can still import plugins
|
|
torch: megatron-core
|
|
torch: .[dev-test]
|
|
|
|
torch_deploy: .[onnx,torch,dev-test]
|
|
commands =
|
|
onnx: python -m pytest tests/unit/onnx
|
|
torch: python -m pytest tests/unit/torch --ignore tests/unit/torch/deploy
|
|
torch_deploy: python -m pytest tests/unit/torch/deploy
|
|
|
|
|
|
###########################################################
|
|
# GPU test environments (Should be used with --current-env)
|
|
###########################################################
|
|
[testenv:cuda13-gpu]
|
|
commands_pre =
|
|
# Install deps here so that it gets installed even in --current-env
|
|
pip install --no-build-isolation git+https://github.com/Dao-AILab/fast-hadamard-transform.git
|
|
pip install -e .[all,dev-test]
|
|
|
|
# Install cupy-cuda13x for INT4 ONNX quantization (default is cupy-cuda12x)
|
|
pip uninstall -y cupy-cuda12x
|
|
pip install cupy-cuda13x
|
|
commands =
|
|
# Coverage fails with "Can't combine line data with arc data" error so not using "--cov"
|
|
python -m pytest tests/gpu
|
|
|
|
[testenv:cuda13-gpu-megatron]
|
|
commands_pre =
|
|
# Install deps here so that it gets installed even in --current-env
|
|
pip install -U megatron-core
|
|
pip install --no-build-isolation git+https://github.com/state-spaces/mamba.git
|
|
pip install -e .[all,dev-test]
|
|
commands =
|
|
# Coverage fails with "Can't combine line data with arc data" error so not using "--cov"
|
|
python -m pytest tests/gpu_megatron
|
|
|
|
#############################################
|
|
# Code quality checks on all files or on diff
|
|
#############################################
|
|
[testenv:{pre-commit}-{all,diff}]
|
|
deps =
|
|
-e .[all,dev-lint]
|
|
commands =
|
|
all: pre-commit run --all-files --show-diff-on-failure {posargs}
|
|
diff: pre-commit run --from-ref origin/main --to-ref HEAD {posargs}
|
|
|
|
|
|
#########################
|
|
# Run documentation build
|
|
#########################
|
|
[testenv:{build,debug}-docs]
|
|
allowlist_externals =
|
|
rm
|
|
passenv =
|
|
SETUPTOOLS_SCM_PRETEND_VERSION
|
|
deps =
|
|
-e .[all,dev-docs]
|
|
changedir = docs
|
|
commands_pre =
|
|
rm -rf build
|
|
rm -rf source/reference/generated
|
|
commands =
|
|
sphinx-build source build/html --fail-on-warning --show-traceback --keep-going
|
|
debug: sphinx-autobuild source build/html --host 0.0.0.0
|
|
|
|
|
|
#################
|
|
# Run wheel build
|
|
#################
|
|
[testenv:build-wheel]
|
|
allowlist_externals =
|
|
bash, cd, rm
|
|
passenv =
|
|
SETUPTOOLS_SCM_PRETEND_VERSION
|
|
deps =
|
|
twine
|
|
commands =
|
|
# Clean build directory to avoid any stale files getting into the wheel
|
|
rm -rf build
|
|
|
|
# Build and check wheel
|
|
pip wheel --no-deps --wheel-dir=dist .
|
|
twine check dist/*
|
|
|
|
# Install and test the wheel
|
|
bash -c "find dist -name 'nvidia_modelopt-*.whl' | xargs pip install -f dist"
|
|
bash -c "cd dist; python -c 'import modelopt; print(modelopt.__version__);'"
|