mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? Type of change: Bug fix + new tests (test-suite only; changes are confined to `tests/`) - **Fix `test_parallel_load_and_export` hang on 2 GPUs.** The temp paths were built from `os.getpid()` *inside* the workers, so each rank got a different `ckpt_dir`; rank 1 then took `_resolve_checkpoint_dir`'s Hub branch and blocked on an extra barrier while rank 0 ran the loader's broadcasts. The checkpoint is now built once in the parent under `tmp_path` and passed in, so all ranks agree on the path. - **Imports moved to module top** across `tests/gpu*` and `tests/examples`; function-local imports kept only where guarded (`importorskip`/`try`), where the import *is* the test (JIT compile), or where it must follow `sys.path` setup. - **Reuse `_test_utils` instead of local copies:** added `get_tiny_mixtral`; deduped `assert_nodes_are_quantized` (5 copies), the accelerate-offload/layerwise config helpers, `make_quant_attention`, `get_dflash_config`, the NVFP4 amax assertions, and 3 copies of the `tiny_wan22_path` fixture. - **Shared model-dir fixtures assert they were not modified** (`assert_unmodified_tree`): a file manifest is compared on teardown, so a test that writes into a session-scoped fixture directory fails instead of silently changing what later tests see. - **Dropped dependency guards the CI env already guarantees** (diffusers, tensorrt_llm in `gpu_trtllm`, transformer_engine in `gpu_megatron`, transformers in examples) so a missing dep fails loudly instead of skipping. - **`test_heterogenous_sharded_state_dict` is skipped on Blackwell** (sm_120), matching the existing marker and its TE/CUDA-13 rationale — same tracking issue as #1901. ### Testing Local, 2x RTX 6000 Ada: `tests/gpu/torch/utils/test_model_load_utils.py` passes on 2 GPUs and on 1 GPU (previously hung on 2). Also ran the touched files in `tests/unit` (608 passed) and `tests/gpu` (~340 passed). ### Before your PR is "*Ready for review*" - Is this change backward compatible?: ✅ - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: N/A - Did you write any new necessary tests?: N/A — test-only PR - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: N/A - Did you get Claude approval on this PR?: ❌ — not yet run <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **Tests** * Added reusable validation utilities for generated files, quantization behavior, attention modules, model fixtures, offloading, and speculative decoding. * Expanded coverage for tiny Wan, Mixtral, Llama, and related model scenarios. * Consolidated duplicated setup and assertions across ONNX, GPU, quantization, export, and sparsity tests. * Improved fixture integrity checks, NVFP4 validation, and handling of identity inputs. * Reduced unnecessary dependency-based skips and isolated known platform-specific flakiness. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com>
182 lines
5.8 KiB
Python
182 lines
5.8 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
import os
|
|
import platform
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
import torch
|
|
import torch.distributed as dist
|
|
from _test_utils.fs_utils import assert_unmodified_tree
|
|
from _test_utils.torch.distributed.utils import init_process
|
|
|
|
import modelopt.torch.opt as mto
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def verbose(request):
|
|
return request.config.getoption("verbose")
|
|
|
|
|
|
def pytest_addoption(parser):
|
|
parser.addoption(
|
|
"--run-manual",
|
|
action="store_true",
|
|
default=False,
|
|
help="Run manual tests",
|
|
)
|
|
parser.addoption(
|
|
"--run-release",
|
|
action="store_true",
|
|
default=False,
|
|
help="Run release tests",
|
|
)
|
|
|
|
|
|
# Default per-test `call` wall-clock cap (seconds) by top-level tests/ subdirectory
|
|
# Every collectible test group must be listed here else collection errors occur
|
|
# A test can override its cap by adding ``@pytest.mark.timeout(...)``
|
|
_DEFAULT_TIMEOUT = {
|
|
"examples": int(os.environ.get("MODELOPT_QA_TEST_TIMEOUT", 300)),
|
|
"gpu": 120,
|
|
"gpu_megatron": 120,
|
|
"gpu_trtllm": 60,
|
|
"gpu_vllm": 60,
|
|
"regression": 180,
|
|
"unit": 120 if platform.system() == "Windows" else 60,
|
|
}
|
|
|
|
|
|
def pytest_collection_modifyitems(config, items):
|
|
"""Skip flag-gated tests and apply a default per-test timeout based on the test directory."""
|
|
skip_marks = [
|
|
("manual", "--run-manual"),
|
|
("release", "--run-release"),
|
|
]
|
|
|
|
for mark_name, option_name in skip_marks:
|
|
if not config.getoption(option_name):
|
|
skipper = pytest.mark.skip(reason=f"Only run when {option_name} is given")
|
|
for item in items:
|
|
if mark_name in item.keywords:
|
|
item.add_marker(skipper)
|
|
|
|
tests_root = Path(__file__).parent
|
|
for item in items:
|
|
if item.get_closest_marker("timeout") is not None or not item.path.is_relative_to(
|
|
tests_root
|
|
):
|
|
continue
|
|
# First path component under tests/ is the group dir (unit, gpu, examples, ...).
|
|
# Crash loudly (rather than silently skip) if a group has no configured default, so a
|
|
# newly added tests/<group>/ must be given an explicit timeout in the mapping above.
|
|
group = item.path.relative_to(tests_root).parts[0]
|
|
if group not in _DEFAULT_TIMEOUT:
|
|
raise pytest.UsageError(
|
|
f"tests/{group}/ has no default timeout; add '{group}' to "
|
|
"_DEFAULT_TIMEOUT in tests/conftest.py."
|
|
)
|
|
item.add_marker(pytest.mark.timeout(_DEFAULT_TIMEOUT[group]))
|
|
|
|
|
|
# General Fixtures #################################################################################
|
|
@pytest.fixture
|
|
def skip_on_windows():
|
|
if platform.system() == "Windows":
|
|
pytest.skip("Skipping on Windows")
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def num_gpus():
|
|
return torch.cuda.device_count()
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def cuda_capability():
|
|
if not torch.cuda.is_available():
|
|
pytest.skip("CUDA is not available")
|
|
return torch.cuda.get_device_capability()
|
|
|
|
|
|
@pytest.fixture
|
|
def distributed_setup_size_1():
|
|
init_process(rank=0, size=1, backend="nccl")
|
|
yield
|
|
dist.destroy_process_group()
|
|
|
|
|
|
@pytest.fixture
|
|
def need_2_gpus():
|
|
if torch.cuda.device_count() < 2:
|
|
pytest.skip("Need at least 2 GPUs to run this test")
|
|
|
|
|
|
@pytest.fixture
|
|
def need_4_gpus():
|
|
if torch.cuda.device_count() < 4:
|
|
pytest.skip("Need at least 4 GPUs to run this test")
|
|
|
|
|
|
@pytest.fixture
|
|
def need_8_gpus():
|
|
if torch.cuda.device_count() < 8:
|
|
pytest.skip("Need at least 8 GPUs to run this test")
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def set_torch_dtype(request):
|
|
orig_dtype = torch.get_default_dtype()
|
|
torch.set_default_dtype(request.param)
|
|
yield
|
|
torch.set_default_dtype(orig_dtype)
|
|
|
|
|
|
@pytest.fixture(scope="session", autouse=True)
|
|
def enable_hf_checkpointing():
|
|
mto.enable_huggingface_checkpointing()
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def project_root_path(request: pytest.FixtureRequest) -> Path:
|
|
"""Fixture providing the project root path for tests."""
|
|
return Path(request.config.rootpath)
|
|
|
|
|
|
# Transformers Models Fixtures #####################################################################
|
|
@pytest.fixture
|
|
def tiny_tokenizer():
|
|
"""Real tiny HF tokenizer (vocab=128) shared across unit and gpu test lanes."""
|
|
# Lazy import: transformers_models.py runs ``pytest.importorskip("transformers")``
|
|
# at module load, which we don't want to trigger at conftest import time.
|
|
from _test_utils.torch.transformers_models import get_tiny_tokenizer
|
|
|
|
return get_tiny_tokenizer()
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def tiny_wan22_path(tmp_path_factory):
|
|
"""Tiny Wan 2.2 pipeline dir, built once per session (the build is the expensive part).
|
|
|
|
Shared by the gpu sparse-attention tests and the diffusers example tests.
|
|
"""
|
|
# Lazy import for the same reason as ``tiny_tokenizer``: diffusers_models.py pulls in
|
|
# transformers at module load.
|
|
from _test_utils.torch.diffusers_models import create_tiny_wan22_pipeline_dir
|
|
|
|
pipeline_dir = create_tiny_wan22_pipeline_dir(tmp_path_factory.mktemp("tiny_wan22"))
|
|
with assert_unmodified_tree(pipeline_dir) as path:
|
|
yield str(path)
|