diff --git a/CHANGELOG.rst b/CHANGELOG.rst index 82e680f56..c1e207c2b 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -83,7 +83,6 @@ Changelog - Remove the deprecated ``examples/llm_autodeploy`` example (deprecated in 0.45). Use TensorRT-LLM's `AutoDeploy `_ directly together with ModelOpt PTQ in ``examples/hf_ptq``. - Remove the deprecated ``examples/llm_qad`` Megatron-LM QAD example (deprecated in 0.45). Use the `megatron_bridge QAD example `_ instead, which provides a simpler Python-based interface and better model coverage. - Dropped VILA / NVILA vision-language model support in ``examples/hf_ptq``. VILA's modeling code requires ``transformers<=4.50.0``, which conflicts with ModelOpt's minimum supported ``transformers`` version. The VILA-specific bootstrap (repo clone, ``requirements-vila.txt``) and loading paths in ``example_utils.py`` have been removed. -- ``examples/hf_ptq`` no longer auto-caps GPU memory for ``device_map="auto"``. It used to build a throwaway meta-device model, run ``infer_auto_device_map`` on it, and -- if the weights would spill to CPU -- shrink ``max_memory`` by ``--gpu_max_mem_percentage`` to leave room for calibration activations. Deciding that requires constructing the model, and remote-code checkpoints that derive scalar hyperparameters from real tensors in ``__init__`` (Phi-4-multimodal's conformer does ``int(torch.tensor(...))``) cannot be built under the global ``torch.device("meta")`` context accelerate uses, so the run died with ``Tensor.item() cannot be called on meta tensors`` before ``from_pretrained`` was ever reached (NVBug 6563509) -- even though both Transformers 4.x and 5.x load those models fine. ``device_map="auto"`` now sizes itself, as ``from_pretrained`` runs its own ``infer_auto_device_map`` regardless. Pass ``--use_seq_device_map`` if quantization hits GPU OOM; ``--gpu_max_mem_percentage`` applies there and to ``--offload_folder``, matching what its help text already documented. - Dropped **Phi-4-multimodal** PTQ support in ``examples/hf_ptq`` (NVBug 6563509). Its bundled remote code predates Transformers v5 and no longer loads on any version ModelOpt supports (``transformers>=4.57``): it requires ``transformers<4.52`` because it reaches ``prepare_inputs_for_generation`` through ``peft``, which needs ``PreTrainedModel`` to still inherit ``GenerationMixin``, and it declares ``_tied_weights_keys`` as a list, which Transformers 5.x rejects. **Phi-3-vision** is dropped alongside it: it is the older, superseded model in the same family, so with its successor unsupportable there is no reason to keep carrying the predecessor. (Phi-3-vision shares the list-valued ``_tied_weights_keys`` defect and so is likewise broken on Transformers 5.x, though it does not hit the ``peft`` blocker.) The support-matrix row, the ``phi4mm`` model type, the multimodal-detection heuristics that only ever matched these two (``vision_lora`` / ``audio_processor`` / ``embd_layer.image_embd_layer``), the ``Phi3Image`` / ``PhiImage`` embedding-export exclusions, and the ``modelopt_recipes/huggingface/phi4mm/`` recipes have been removed. Text-only Phi-3/Phi-4 and Phi-3.5-MoE are unaffected. **Deprecations** diff --git a/examples/hf_ptq/example_utils.py b/examples/hf_ptq/example_utils.py index 9018d267e..34fa5a975 100755 --- a/examples/hf_ptq/example_utils.py +++ b/examples/hf_ptq/example_utils.py @@ -33,6 +33,7 @@ from typing import Any import torch import transformers import yaml +from accelerate import infer_auto_device_map, init_empty_weights from accelerate.utils import get_max_memory from safetensors import safe_open from transformers import ( @@ -637,6 +638,25 @@ def get_original_hf_quant_method(config) -> str | None: return None +def _resolve_init_config(hf_config, auto_model_module, ckpt_path, config_kwargs): + """Re-derive a built-in config when a remote-code config is used with a built-in model + class, so it matches the model definition's version; fall back to hf_config otherwise. + """ + if auto_model_module in [AutoModelForCausalLM, AutoModel]: + return hf_config + if not type(hf_config).__module__.startswith("transformers_modules"): + return hf_config + builtin_config_kwargs = {k: v for k, v in config_kwargs.items() if k != "trust_remote_code"} + try: + return AutoConfig.from_pretrained(ckpt_path, **builtin_config_kwargs) + except Exception as e: + warnings.warn( + f"Could not re-derive a built-in config for {ckpt_path} ({e}); using the " + "remote-code config for device-map inference." + ) + return hf_config + + def _get_config_dtype(config): config_dtype = ( getattr(config, "dtype", None) or getattr(config, "torch_dtype", None) or torch.bfloat16 @@ -838,19 +858,30 @@ def get_model( auto_model_module = AutoModel else: auto_model_module = AutoModelForCausalLM + from_config = auto_model_module.from_config else: auto_model_module = getattr(transformers, architecture) + from_config = auto_model_module._from_config - # Assume bfloat16 precision by default, unless specified by the hf_config. - config_dtype = _get_config_dtype(hf_config) + config_for_init = _resolve_init_config( + hf_config, auto_model_module, ckpt_path, config_kwargs + ) + + with init_empty_weights(include_buffers=True): + # When computing the device_map, assuming bfloat16 precision by default, + # unless specified by the hf_config. + config_dtype = _get_config_dtype(config_for_init) + model_kwargs2 = _apply_dtype_to_config( + model_kwargs, config_dtype, architecture, apply_config_dtype=True + ) + if auto_model_module not in [AutoModelForCausalLM, AutoModel]: + model_kwargs2.pop("trust_remote_code", None) + model_kwargs2.pop("max_memory", None) + model = from_config(config_for_init, **model_kwargs2) + + max_memory = get_max_memory() - # device_map="auto" is left to size itself. Capping it here needs to know whether - # the weights fit, which needs a constructed model, whose __init__ some remote-code - # checkpoints cannot survive on a meta device (NVBug 6563509) -- and capping when - # they *do* fit would introduce CPU offload that was not there. --use_seq_device_map - # is the supported answer to GPU OOM, and --gpu_max_mem_percentage applies there. if _disk_offload: - max_memory = get_max_memory() for _k in max_memory: if isinstance(_k, int): if max_gpu_memory_gb is not None: @@ -866,6 +897,20 @@ def get_model( f"Offload folder: {offload_folder}\n" "Weights exceeding GPU+CPU budgets will be streamed from disk." ) + else: + inferred_device_map = infer_auto_device_map(model, max_memory=max_memory) + if "cpu" in inferred_device_map.values(): + for _device in max_memory: + if isinstance(_device, int): + max_memory[_device] *= gpu_mem_percentage + + print( + "Model does not fit to the GPU mem. " + f"We apply the following memory limit for calibration: \n{max_memory}\n" + "If you hit GPU OOM issue, please adjust `gpu_mem_percentage` or " + "reduce the calibration `batch_size` manually." + ) + model_kwargs["max_memory"] = max_memory model_kwargs2 = _apply_dtype_to_config(model_kwargs, config_dtype, architecture) if _disk_offload: diff --git a/tests/examples/hf_ptq/test_example_utils.py b/tests/examples/hf_ptq/test_example_utils.py index 66b853e3b..4b9120b04 100644 --- a/tests/examples/hf_ptq/test_example_utils.py +++ b/tests/examples/hf_ptq/test_example_utils.py @@ -19,7 +19,9 @@ separate-file-standalone, separate-file-indexed) plus a negative case. """ import json +from contextlib import nullcontext from types import SimpleNamespace +from unittest.mock import patch import pytest import torch @@ -197,15 +199,59 @@ def test_get_original_hf_quant_method_none_for_unquantized(): ) +# ---------- _resolve_init_config --------------------------------------------- + + +def _remote_config(): + # Config whose class module lives under "transformers_modules" (remote code). + cls = type("_RemoteConfig", (), {"__module__": "transformers_modules.ckpt.config"}) + return cls() + + +def test_resolve_init_config_rederives_for_remote_config(): + builtin_cfg = SimpleNamespace() + with patch.object( + example_utils.AutoConfig, "from_pretrained", return_value=builtin_cfg + ) as mock: + out = example_utils._resolve_init_config( + _remote_config(), object, "/ckpt", {"trust_remote_code": True} + ) + assert out is builtin_cfg + mock.assert_called_once_with("/ckpt") # trust_remote_code stripped + + +def test_resolve_init_config_keeps_non_remote_config(): + cfg = SimpleNamespace() # module is "types", not remote + with patch.object(example_utils.AutoConfig, "from_pretrained") as mock: + assert example_utils._resolve_init_config(cfg, object, "/ckpt", {}) is cfg + mock.assert_not_called() + + +def test_resolve_init_config_falls_back_when_rederive_raises(): + cfg = _remote_config() + with patch.object(example_utils.AutoConfig, "from_pretrained", side_effect=ValueError()): + assert example_utils._resolve_init_config(cfg, object, "/ckpt", {}) is cfg + + @pytest.mark.parametrize( - ("architecture", "model_class_name"), + ( + "architecture", + "model_class_name", + "expected_config_dtype_kwarg", + "unexpected_config_dtype_kwarg", + ), [ - # DeciLM takes the legacy ``torch_dtype`` kwarg; everything else takes ``dtype``. - ("DeciLMForCausalLM", "AutoModelForCausalLM"), - ("LlamaForCausalLM", "LlamaForCausalLM"), + ("DeciLMForCausalLM", "AutoModelForCausalLM", "torch_dtype", "dtype"), + ("LlamaForCausalLM", "LlamaForCausalLM", "dtype", "torch_dtype"), ], ) -def test_get_model_uses_expected_dtype_kwarg(monkeypatch, architecture, model_class_name): +def test_get_model_uses_expected_dtype_kwarg( + monkeypatch, + architecture, + model_class_name, + expected_config_dtype_kwarg, + unexpected_config_dtype_kwarg, +): calls = {} hf_config = SimpleNamespace( architectures=[architecture], @@ -221,7 +267,12 @@ def test_get_model_uses_expected_dtype_kwarg(monkeypatch, architecture, model_cl class FakeAutoModelForCausalLM: @staticmethod def from_config(config, **kwargs): - raise AssertionError("get_model must not construct a model to size the load") + calls["from_config"] = kwargs + assert config is hf_config + assert kwargs[expected_config_dtype_kwarg] is torch.float16 + assert unexpected_config_dtype_kwarg not in kwargs + assert "max_memory" not in kwargs + return FakeModel() @staticmethod def from_pretrained(*args, **kwargs): @@ -252,12 +303,18 @@ def test_get_model_uses_expected_dtype_kwarg(monkeypatch, architecture, model_cl monkeypatch.setattr(example_utils.transformers, model_class_name, FakeLlamaForCausalLM) monkeypatch.setattr(example_utils, "is_nemotron_vl", lambda config: False) monkeypatch.setattr(example_utils, "is_speculative", lambda config: False) + monkeypatch.setattr(example_utils, "init_empty_weights", lambda include_buffers: nullcontext()) monkeypatch.setattr(example_utils, "get_max_memory", lambda: {0: 1024}) + monkeypatch.setattr(example_utils, "infer_auto_device_map", lambda model, max_memory: {"": 0}) model = example_utils.get_model("checkpoint", device="cpu", trust_remote_code=True) assert isinstance(model, FakeModel) assert calls["eval"] + if expected_config_dtype_kwarg == "torch_dtype": + assert calls["from_config"]["trust_remote_code"] is True + else: + assert "trust_remote_code" not in calls["from_config"] assert calls["from_pretrained"]["trust_remote_code"] is True @@ -312,7 +369,9 @@ def test_get_model_device_map_for_diffusion_gemma( monkeypatch.setattr(example_utils.transformers, architecture, FakeArchitecture, raising=False) monkeypatch.setattr(example_utils, "is_nemotron_vl", lambda config: False) monkeypatch.setattr(example_utils, "is_speculative", lambda config: False) + monkeypatch.setattr(example_utils, "init_empty_weights", lambda include_buffers: nullcontext()) monkeypatch.setattr(example_utils, "get_max_memory", lambda: {0: 1024}) + monkeypatch.setattr(example_utils, "infer_auto_device_map", lambda model, max_memory: {"": 0}) monkeypatch.setattr(torch.cuda, "device_count", lambda: device_count) example_utils.get_model("checkpoint", device="cuda", trust_remote_code=True) @@ -405,24 +464,10 @@ def test_get_model_deepseek_honors_trust_remote_code( ) monkeypatch.setattr(example_utils, "is_nemotron_vl", lambda config: False) monkeypatch.setattr(example_utils, "is_speculative", lambda config: False) + monkeypatch.setattr(example_utils, "init_empty_weights", lambda include_buffers: nullcontext()) monkeypatch.setattr(example_utils, "get_max_memory", lambda: {0: 1024}) + monkeypatch.setattr(example_utils, "infer_auto_device_map", lambda model, max_memory: {"": 0}) example_utils.get_model("checkpoint", device="cpu", trust_remote_code=trust_remote_code) assert used["path"] == ("bundled" if expect_bundled_code else "builtin") - - -def _patch_get_model_deps(monkeypatch, hf_config, model_class): - monkeypatch.setattr(example_utils.AutoConfig, "from_pretrained", lambda *a, **kw: hf_config) - monkeypatch.setattr(example_utils.transformers, "LlamaForCausalLM", model_class) - monkeypatch.setattr(example_utils, "is_nemotron_vl", lambda config: False) - monkeypatch.setattr(example_utils, "is_speculative", lambda config: False) - - -def _llama_config(): - return SimpleNamespace( - architectures=["LlamaForCausalLM"], - dtype=torch.float16, - model_type="llama", - torch_dtype=torch.bfloat16, - )