mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? Type of change: Bug fix (CI / tests) - **Fix Nemotron nightly failures:** install `mamba_ssm`/`causal-conv1d` from PyPI releases instead of git `main` (avoids the broken `apache-tvm-ffi 0.1.12` that crashes on import). - **Speed up CUDA builds:** set `TORCH_CUDA_ARCH_LIST=12.0` (runner's sm_120) in the GPU/example/regression workflow container env instead of the image's ~6 archs. - **Make unit tests CPU-only:** force CUDA off in the nox `unit` env and skip JIT-compiling CUDA extensions when no GPU is usable; move the two GPU-/`mamba_ssm`-requiring unit tests to `tests/gpu`. - **Harden example tests against HF flakes:** capture subprocess output and retry transient HuggingFace access errors (5xx / rate-limit / connection). - **Skip Blackwell-flaky sharded-state-dict tests:** `test_homogeneous_sharded_state_dict` and `test_regular_state_dict[320]` intermittently hit a CUDA illegal-memory-access on the sm_120 runner that poisons the CUDA context and cascades timeouts; gate them behind a reusable `skip_flaky_on_blackwell` marker (still run on non-Blackwell GPUs). - **Bump slow test timeout:** `test_prune_minitron_vlm` → 360s for the 2-GPU nightly. ### Testing - CI tests on this PR pass (1-gpu) - Manually triggerred 2-gpu test: - GPU: https://github.com/NVIDIA/Model-Optimizer/actions/runs/28774553356 - Examples: https://github.com/NVIDIA/Model-Optimizer/actions/runs/28774556693 - Regression: https://github.com/NVIDIA/Model-Optimizer/actions/runs/28771197586 ### Additional Information - Backward compatible: N/A (CI/tests only) - New dependency: N/A - Changelog: N/A (CI/test infra) 🤖 Generated with [Claude Code](https://claude.com/claude-code) --------- Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com> Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
97 lines
4.0 KiB
Python
97 lines
4.0 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""Utility functions for loading CPP / CUDA extensions."""
|
|
|
|
import os
|
|
import warnings
|
|
from pathlib import Path
|
|
from time import time
|
|
from types import ModuleType
|
|
from typing import Any
|
|
|
|
import torch
|
|
from packaging.specifiers import SpecifierSet
|
|
from packaging.version import Version
|
|
from torch.utils.cpp_extension import load
|
|
|
|
__all__ = ["load_cpp_extension"]
|
|
|
|
|
|
def load_cpp_extension(
|
|
name: str,
|
|
sources: list[str | Path],
|
|
cuda_version_specifiers: str | None,
|
|
fail_msg: str = "",
|
|
raise_if_failed: bool = False,
|
|
**load_kwargs: Any,
|
|
) -> ModuleType | None:
|
|
"""Load a C++ / CUDA extension using torch.utils.cpp_extension.load() if the current CUDA version satisfies it.
|
|
|
|
Loading first time may take a few mins because of the compilation, but subsequent loads are instantaneous.
|
|
|
|
Args:
|
|
name: Name of the extension.
|
|
sources: Source files to compile.
|
|
cuda_version_specifiers: Specifier (e.g. ">=11.8,<12") for CUDA versions required to enable the extension.
|
|
fail_msg: Additional message to display if the extension fails to load.
|
|
raise_if_failed: Raise an exception if the extension fails to load.
|
|
**load_kwargs: Keyword arguments to torch.utils.cpp_extension.load().
|
|
"""
|
|
ext = None
|
|
print(f"Loading extension {name}...")
|
|
start = time()
|
|
|
|
if torch.version.cuda is None or not torch.cuda.is_available():
|
|
fail_msg = f"Skipping extension {name} because CUDA is not available."
|
|
elif cuda_version_specifiers and Version(torch.version.cuda) not in SpecifierSet(
|
|
cuda_version_specifiers
|
|
):
|
|
fail_msg = (
|
|
f"Skipping extension {name} because the current CUDA version {torch.version.cuda}"
|
|
f" does not satisfy the specifiers {cuda_version_specifiers}."
|
|
)
|
|
else:
|
|
if not os.environ.get("TORCH_CUDA_ARCH_LIST"):
|
|
device_capability = torch.cuda.get_device_capability()
|
|
os.environ["TORCH_CUDA_ARCH_LIST"] = f"{device_capability[0]}.{device_capability[1]}"
|
|
if os.name == "nt":
|
|
# Define USE_CUDA so PyTorch's compiled_autograd.h takes its Windows-safe branch;
|
|
# otherwise, nvcc + MSVC fail with "error C2872: 'std': ambiguous symbol".
|
|
# See https://github.com/pytorch/pytorch/issues/148317
|
|
for key in ("extra_cflags", "extra_cuda_cflags"):
|
|
flags = list(load_kwargs.get(key, []))
|
|
if not any("USE_CUDA" in flag for flag in flags):
|
|
flags.append("-DUSE_CUDA=1")
|
|
load_kwargs[key] = flags
|
|
try:
|
|
ext = load(name, sources, **load_kwargs)
|
|
except Exception as e:
|
|
if not fail_msg:
|
|
fail_msg = f"Unable to load extension {name} and falling back to CPU version."
|
|
fail_msg = f"{e}\n{fail_msg}"
|
|
# RuntimeError can be raised if there are any errors while compiling the extension.
|
|
# OSError can be raised if CUDA_HOME path is not set correctly.
|
|
# subprocess.CalledProcessError can be raised on `-runtime` images where c++ is not installed.
|
|
|
|
if ext is None:
|
|
if raise_if_failed:
|
|
raise RuntimeError(fail_msg)
|
|
else:
|
|
warnings.warn(fail_msg)
|
|
else:
|
|
print(f"Loaded extension {name} in {time() - start:.1f} seconds")
|
|
return ext
|