mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? Type of change: Bug fix (CI) + test coverage **Fixes `onnx (torch_onnx)` and `onnx (diffusers)`**, which have failed on every branch since `torch 2.14.0` was published to PyPI today (2026-09-02 13:42 UTC), and **adds torch 2.14 to the unit test matrix as the new default** so the next torch release is caught there rather than in an example job. ### Root cause Every test in those two jobs failed with: ``` RuntimeError: CUDNN_BACKEND_TENSOR_DESCRIPTOR cudnnFinalize failed ptrDesc->finalize() cudnn_status: CUDNN_STATUS_SUBLIBRARY_LOADING_FAILED ``` `nvcr.io/nvidia/tensorrt:26.05-py3` ships cuDNN **9.22** and has no preinstalled torch, so pip resolved the newest one — and torch 2.14 pins `nvidia-cudnn-cu13==9.24.0.43`. Loading 9.24 sublibraries against the image's 9.22 `libcudnn.so.9` is exactly what that status reports. | | last good run (08:55) | first failing run (13:34) | |---|---|---| | `torch` | 2.13.0 | **2.14.0** | | `nvidia-cudnn-cu13` | 9.20.0.48 | **9.24.0.43** | | image cuDNN | 9.22.0.52 | 9.22.0.52 | ### Why only these two jobs - The **nemo** and **pytorch** images have a preinstalled torch that already satisfies `torch>=2.8`, so pip never resolves a new one — confirmed from the megatron job log, where torch does not appear in `Successfully installed`. - **`tensorrt:26.05-py3` has no preinstalled torch**, so pip takes the newest from PyPI. - **`onnx (torch_trt)`** shares that image but passes throughout, because `torch-tensorrt<2.13` already holds torch below 2.14. ### The changes 1. **Constrain torch only where the incompatibility is.** `PIP_CONSTRAINT=torch<2.14` in the example runner, applied when the job's image is a `tensorrt` one. It also covers the `examples/*/requirements.txt` loop in the same shell, which matters because `nemo_automodel` pulls torch in too. Not pinned in `pyproject.toml`: torch 2.14 is fine anywhere its own bundled cuDNN is the one loaded, so that would constrain users to work around one pinned image. 2. **Test torch 2.14.** `torch_214` added to `TORCH_VERSIONS` (`torchvision~=0.29.0`) and promoted to the unit-test default across the supported Python versions, with 2.13 demoted to the back-compat row. `release.yml`'s basic unit test moves to the same default (it was still on 2.12). Nothing exercised 2.14 before — which is why a torch release reached us through an example job instead of a unit test. ### Testing - `actionlint` and YAML/TOML parse clean; pre-commit clean. - Verified by this PR's own jobs: `onnx (torch_onnx)` and `onnx (diffusers)` reproduce the failure on `main` right now, and the new `unit-3.12(torch_214, tf_latest)` job is the first run of ModelOpt against torch 2.14. ### Before your PR is "*Ready for review*" - Is this change backward compatible?: ✅ — CI-only; no source or package metadata change - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: N/A — no new dependency - Did you write any new necessary tests?: ✅ — torch 2.14 added to the unit test matrix - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: N/A — internal CI, not user-facing - Did you get Claude approval on this PR?: ❌ — not yet requested --------- Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com>
222 lines
9.5 KiB
Python
222 lines
9.5 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
"""Nox session definitions for testing, linting, docs, and wheel builds.
|
|
|
|
Usage:
|
|
python -m pip install nox uv # install nox and uv (once)
|
|
nox -l # list all sessions
|
|
nox -s gpu_megatron # run a GPU session (inside container)
|
|
nox -s "unit-3.12(torch_211, tf_latest)" # run a specific unit test combination
|
|
nox -s "unit-3.12(torch_211, tf_latest)" -R # force-recreate venv (e.g. after dep changes)
|
|
COVERAGE_PROCESS_START=pyproject.toml nox -s "unit-3.12(torch_211, tf_latest)" # with coverage
|
|
"""
|
|
|
|
import glob
|
|
import os
|
|
import shutil
|
|
|
|
import nox
|
|
|
|
nox.options.default_venv_backend = "uv" if shutil.which("uv") else "virtualenv"
|
|
nox.options.envdir = "/tmp/.nox"
|
|
nox.options.reuse_existing_virtualenvs = True
|
|
|
|
TORCH_VERSIONS = {
|
|
"torch_28": "torchvision~=0.23.0",
|
|
"torch_29": "torchvision~=0.24.0",
|
|
"torch_210": "torchvision~=0.25.0",
|
|
"torch_211": "torchvision~=0.26.0",
|
|
"torch_212": "torchvision~=0.27.0",
|
|
"torch_213": "torchvision~=0.28.0",
|
|
"torch_214": "torchvision~=0.29.0",
|
|
}
|
|
|
|
# Extra install pins applied per transformers matrix entry (installed after the base
|
|
# ``.[all,dev-test]`` install to constrain that env).
|
|
TRANSFORMERS_VERSIONS = {
|
|
"tf_latest": ("transformers~=5.14.0",),
|
|
# transformers 4.57 caps ``huggingface_hub<1.0``, but ``diffusers>=0.40`` requires
|
|
# ``huggingface_hub>=1.23``. Bound diffusers to a hub<1.0-compatible release so this env
|
|
# stays internally consistent; otherwise diffusers' pipeline import fails and diffusers
|
|
# models silently misroute to the LLM path on export.
|
|
"tf_min": ("transformers~=4.57.0", "diffusers<0.40"),
|
|
}
|
|
|
|
|
|
def _cov_args():
|
|
"""Return --cov when COVERAGE_PROCESS_START is set (CI only)."""
|
|
return ["--cov"] if os.environ.get("COVERAGE_PROCESS_START") else []
|
|
|
|
|
|
# ─── CPU unit tests ───────────────────────────────────────────────────────────
|
|
_CPU_ONLY_ENV = {"CUDA_VISIBLE_DEVICES": ""}
|
|
|
|
|
|
@nox.session(python=["3.10", "3.11", "3.12", "3.13", "3.14"])
|
|
@nox.parametrize("tf_ver", [nox.param(k, id=k) for k in TRANSFORMERS_VERSIONS])
|
|
@nox.parametrize("torch_ver", [nox.param(k, id=k) for k in TORCH_VERSIONS])
|
|
def unit(session, torch_ver, tf_ver):
|
|
"""Unit tests — parametrized over torch and transformers versions."""
|
|
session.install(TORCH_VERSIONS[torch_ver], "-e", ".[all,dev-test]")
|
|
tf_pins = TRANSFORMERS_VERSIONS[tf_ver]
|
|
if tf_pins:
|
|
session.install(*tf_pins)
|
|
session.run("python", "-m", "pytest", "tests/unit", *_cov_args(), env=_CPU_ONLY_ENV)
|
|
|
|
|
|
@nox.session(python="3.12")
|
|
@nox.parametrize("subset", ["onnx", "torch", "torch_deploy"])
|
|
def partial_unit(session, subset):
|
|
"""Unit tests with partial installs."""
|
|
if subset == "onnx":
|
|
session.install("torchvision~=0.26.0", ".[onnx,dev-test]")
|
|
session.run("python", "-m", "pytest", "tests/unit/onnx", env=_CPU_ONLY_ENV)
|
|
elif subset == "torch":
|
|
session.install("megatron-core", ".[dev-test]")
|
|
session.run(
|
|
"python",
|
|
"-m",
|
|
"pytest",
|
|
"tests/unit/torch",
|
|
"--ignore=tests/unit/torch/deploy",
|
|
"--ignore=tests/unit/torch/puzzletron",
|
|
env=_CPU_ONLY_ENV,
|
|
)
|
|
else: # torch_deploy
|
|
session.install(".[onnx,dev-test]")
|
|
session.run("python", "-m", "pytest", "tests/unit/torch/deploy", env=_CPU_ONLY_ENV)
|
|
|
|
|
|
# ─── GPU sessions (run inside containers — no new venv) ──────────────────────
|
|
# `venv_backend="none"` skips creating a new venv so the session runs directly in the container's
|
|
# existing Python environment (e.g. /opt/venv in NeMo) instead of an isolated one.
|
|
# Use `python -m pip/pytest` to ensure the container's active venv Python is used,
|
|
# not a stale PATH entry (e.g. NeMo container has pip → /usr/local/bin/pip but python → /opt/venv/bin/python).
|
|
# Container: nvcr.io/nvidia/pytorch:26.01-py3 or later
|
|
@nox.session(venv_backend="none")
|
|
def gpu(session):
|
|
# tests/gpu/_extensions/test_onnx_extensions.py fails for newer containers
|
|
# until https://github.com/tbenthompson/cppimport/pull/98
|
|
session.run(
|
|
"python",
|
|
"-m",
|
|
"pip",
|
|
"install",
|
|
"--no-build-isolation",
|
|
"git+https://github.com/Dao-AILab/fast-hadamard-transform.git",
|
|
)
|
|
session.run("python", "-m", "pip", "install", "-e", ".[all,dev-test]")
|
|
session.run("python", "-m", "pip", "uninstall", "-y", "cupy-cuda12x")
|
|
session.run("python", "-m", "pip", "install", "cupy-cuda13x")
|
|
session.run(
|
|
"python",
|
|
"-m",
|
|
"pip",
|
|
"install",
|
|
"--no-build-isolation",
|
|
# Install the latest *released* sdists (built against the container torch)
|
|
"mamba_ssm",
|
|
"causal-conv1d",
|
|
)
|
|
session.run("python", "-m", "pytest", "tests/gpu", *_cov_args())
|
|
|
|
|
|
# Container: nvcr.io/nvidia/nemo:26.08 or later
|
|
@nox.session(venv_backend="none")
|
|
def gpu_megatron(session):
|
|
# NeMo containers have transformers 5.x but a system-wide installed trtllm which does not support it causing import errors
|
|
session.run("pip", "uninstall", "-y", "tensorrt_llm")
|
|
# Pre-installed nvidia-modelopt shadows the editable install
|
|
session.run("pip", "uninstall", "-y", "nvidia-modelopt")
|
|
session.run("python", "-m", "pip", "install", "-e", ".[hf,dev-test]")
|
|
session.run("python", "-m", "pytest", "tests/gpu_megatron", *_cov_args())
|
|
|
|
|
|
# Container: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc10 or later
|
|
@nox.session(venv_backend="none")
|
|
def gpu_trtllm(session):
|
|
session.run("python", "-m", "pip", "install", "-e", ".[hf,dev-test]")
|
|
session.run("python", "-m", "pytest", "tests/gpu_trtllm", *_cov_args())
|
|
|
|
|
|
# Container: docker.io/vllm/vllm-openai (the published image ships vLLM + CUDA + torch).
|
|
# Pin must stay in sync with examples/vllm_serve/Dockerfile.
|
|
@nox.session(venv_backend="none")
|
|
def gpu_vllm(session):
|
|
session.run("python3", "-m", "pip", "install", "-e", ".[hf,puzzletron,dev-test]")
|
|
session.run("python3", "-m", "pytest", "tests/gpu_vllm", *_cov_args())
|
|
|
|
|
|
# Container: nvcr.io/nvidia/pytorch:26.01-py3 or later
|
|
@nox.session(venv_backend="none")
|
|
def regression(session):
|
|
session.run("python", "-m", "pip", "install", "-e", ".[hf,dev-test]")
|
|
session.run("python", "-m", "pytest", "tests/regression", *_cov_args())
|
|
|
|
|
|
# ─── Code quality ─────────────────────────────────────────────────────────────
|
|
@nox.session
|
|
def pre_commit_all(session):
|
|
session.install("-e", ".[all,dev-lint]")
|
|
session.run("pre-commit", "run", "--all-files", "--show-diff-on-failure")
|
|
|
|
|
|
@nox.session
|
|
def pre_commit_diff(session):
|
|
session.install("-e", ".[all,dev-lint]")
|
|
session.run("pre-commit", "run", "--from-ref", "origin/main", "--to-ref", "HEAD")
|
|
|
|
|
|
# ─── Docs ─────────────────────────────────────────────────────────────────────
|
|
@nox.session
|
|
def docs(session):
|
|
session.install("-e", ".[all,dev-docs]")
|
|
shutil.rmtree("docs/build", ignore_errors=True)
|
|
shutil.rmtree("docs/source/reference/generated", ignore_errors=True)
|
|
with session.chdir("docs"):
|
|
session.run(
|
|
"sphinx-build",
|
|
"-d",
|
|
"/tmp/doctrees",
|
|
"source",
|
|
"build/html",
|
|
"--fail-on-warning",
|
|
"--show-traceback",
|
|
"--keep-going",
|
|
)
|
|
|
|
|
|
@nox.session
|
|
def docs_debug(session):
|
|
session.install("-e", ".[all,dev-docs]")
|
|
shutil.rmtree("docs/build", ignore_errors=True)
|
|
shutil.rmtree("docs/source/reference/generated", ignore_errors=True)
|
|
with session.chdir("docs"):
|
|
session.run("sphinx-autobuild", "source", "build/html", "--host", "0.0.0.0")
|
|
|
|
|
|
# ─── Wheel build ──────────────────────────────────────────────────────────────
|
|
@nox.session
|
|
def build_wheel(session):
|
|
shutil.rmtree("build", ignore_errors=True)
|
|
session.install("twine")
|
|
session.run("pip", "wheel", "--no-deps", "--wheel-dir=dist", ".")
|
|
wheels = glob.glob("dist/*.whl")
|
|
session.run("twine", "check", *wheels)
|
|
(modelopt_wheel,) = glob.glob("dist/nvidia_modelopt-*.whl")
|
|
session.install(modelopt_wheel, "-f", "dist")
|
|
with session.chdir("dist"):
|
|
session.run("python", "-c", "import modelopt; print(modelopt.__version__)")
|