mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
## What does this PR do? - Code refactor, no logic change - De-couple Minitron pruning importance estimator from Dynamic Module so its easy to configure different importance logic for pruning ## Testing <!-- Mention how have you tested your change if applicable. --> - [x] CI/CD tests passing - [x] Compare mmlu on pruned Qwen3-8B with previous and current implementation - [x] Compare mmlu on pruned Qwen3-30B-A3B with previous and current implementation (lot of variance in results, some pruning configs better some worse) - [x] Compare mmlu on pruned Nemotron-Nano-v2-9B with previous and current implementation Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com>
132 lines
3.9 KiB
INI
132 lines
3.9 KiB
INI
[tox]
|
|
envlist=
|
|
pre-commit-all
|
|
py312-torch28-tf_latest-unit
|
|
py312-cuda12-gpu
|
|
skipsdist = True
|
|
toxworkdir = /tmp/{env:USER}-modelopt-tox
|
|
|
|
|
|
############################
|
|
# CPU Unit test environments
|
|
############################
|
|
[testenv:{py310,py311,py312}-torch{26,27,28,29}-tf_{min,latest}-unit]
|
|
deps =
|
|
# torch version auto-selected based on torchvision version
|
|
torch26: torchvision~=0.21.0
|
|
torch27: torchvision~=0.22.0
|
|
torch28: torchvision~=0.23.0
|
|
torch29: torchvision~=0.24.0
|
|
|
|
# Install megatron-core for special unit tests
|
|
megatron-core
|
|
|
|
-e .[all,dev-test]
|
|
|
|
# Should match setup.py
|
|
tf_min: transformers~=4.53.0
|
|
commands =
|
|
python -m pytest tests/unit {env:COV_ARGS:}
|
|
|
|
|
|
#####################################################################
|
|
# Environment to run unit tests with subset of dependencies installed
|
|
#####################################################################
|
|
[testenv:{py310,py311,py312}-partial-unit-{onnx,torch,torch_deploy}]
|
|
allowlist_externals =
|
|
bash, rm
|
|
deps =
|
|
# Make sure torch 2.9 is used
|
|
torchvision~=0.24.0
|
|
|
|
# ONNX unit tests heavily rely on torch / torchvision
|
|
onnx: .[onnx,dev-test]
|
|
onnx: torchvision
|
|
|
|
# Install megatron-core to test torch-only install can still import plugins
|
|
torch: megatron-core
|
|
torch: .[dev-test]
|
|
|
|
torch_deploy: .[onnx,torch,dev-test]
|
|
commands =
|
|
onnx: python -m pytest tests/unit/onnx
|
|
torch: python -m pytest tests/unit/torch --ignore tests/unit/torch/deploy
|
|
torch_deploy: python -m pytest tests/unit/torch/deploy
|
|
|
|
|
|
###########################################################
|
|
# GPU test environments (Should be used with --current-env)
|
|
###########################################################
|
|
[testenv:{py310,py311,py312}-cuda12-gpu]
|
|
commands_pre =
|
|
# Install deps here so that it gets installed even in --current-env
|
|
pip install -U megatron-core
|
|
pip install git+https://github.com/Dao-AILab/fast-hadamard-transform.git
|
|
|
|
# Skip triton because pytorch-triton is installed in the NGC PyTorch containers
|
|
pip install pip-mark-installed
|
|
pip-mark-installed triton
|
|
pip install --no-build-isolation git+https://github.com/state-spaces/mamba.git
|
|
|
|
# Install Eagle-3 test dependencies
|
|
pip install tiktoken blobfile sentencepiece
|
|
|
|
# NOTE: User is expected to have correct torch-cuda version pre-installed if using --current-env
|
|
# to avoid possible CUDA version mismatch
|
|
pip install -e .[all,dev-test]
|
|
commands =
|
|
# Coverage fails with "Can't combine line data with arc data" error so not using "--cov"
|
|
python -m pytest tests/gpu
|
|
|
|
#############################################
|
|
# Code quality checks on all files or on diff
|
|
#############################################
|
|
[testenv:{pre-commit}-{all,diff}]
|
|
deps =
|
|
-e .[all,dev-lint]
|
|
commands =
|
|
all: pre-commit run --all-files --show-diff-on-failure {posargs}
|
|
diff: pre-commit run --from-ref origin/main --to-ref HEAD {posargs}
|
|
|
|
|
|
#########################
|
|
# Run documentation build
|
|
#########################
|
|
[testenv:{build,debug}-docs]
|
|
allowlist_externals =
|
|
rm
|
|
passenv =
|
|
SETUPTOOLS_SCM_PRETEND_VERSION
|
|
deps =
|
|
-e .[all,dev-docs]
|
|
changedir = docs
|
|
commands_pre =
|
|
rm -rf build
|
|
rm -rf source/reference/generated
|
|
commands =
|
|
sphinx-build source build/html --fail-on-warning --show-traceback --keep-going
|
|
debug: sphinx-autobuild source build/html --host 0.0.0.0
|
|
|
|
|
|
#################
|
|
# Run wheel build
|
|
#################
|
|
[testenv:build-wheel]
|
|
allowlist_externals =
|
|
bash, cd, rm
|
|
passenv =
|
|
SETUPTOOLS_SCM_PRETEND_VERSION
|
|
deps =
|
|
twine
|
|
commands =
|
|
# Clean build directory to avoid any stale files getting into the wheel
|
|
rm -rf build
|
|
|
|
# Build and check wheel
|
|
pip wheel --no-deps --wheel-dir=dist .
|
|
twine check dist/*
|
|
|
|
# Install and test the wheel
|
|
bash -c "find dist -name 'nvidia_modelopt-*.whl' | xargs pip install -f dist"
|
|
bash -c "cd dist; python -c 'import modelopt; print(modelopt.__version__);'"
|