diff --git a/CHANGELOG.rst b/CHANGELOG.rst index ea80a8be0..2cc6b5980 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -40,6 +40,7 @@ Changelog *Misc* - A tracked ``examples/hf_ptq/hf_ptq.py`` run now writes ``.experiment.json`` into ``--export_path`` and uploads the same file with the run, so a checkpoint on disk names the experiment and MLflow run id that produced it. The pointer is written only once the export completes, and an export that is not tracked removes one it would otherwise inherit from a reused ``--export_path`` or from a quantized source checkpoint. +- ``export_hf_checkpoint`` and the vLLM fake-quant export now carry every non-model file of the source checkpoint -- everything but the weights and the metadata the export writes -- into the export verbatim, subdirectories included, instead of regenerating tokenizer and processor files. They read the source from local disk only, so get a local copy of a Hugging Face Hub model first with ``modelopt.torch.export.ensure_local_checkpoint`` -- which, under ``torch.distributed``, downloads once on rank 0 of the default group or of the ``group`` passed -- and load it from that directory. **Backward Breaking Changes** @@ -90,6 +91,7 @@ Changelog **Bug Fixes** +- Fix Hugging Face exports of checkpoints with off-index safetensors (such as GLM-4.7's ``mtp.safetensors``) missing those files: ``export_hf_checkpoint`` now writes them itself. - Fix Megatron-Core checkpoint saving for quantized grouped MoE experts when tensor and expert parallelism are both enabled. - Fix shared ONNX export metadata and Diffusers attention policy: every ``NVFP4QuantExporter`` post-process now upgrades the default-domain opset to at least 23, all FP8 custom-op exports re-run ONNX shape/type inference after setting output metadata, and quantized SDPA derives FP8 MHA enablement from the live Q/K/V quantizers instead of honoring a caller-set ``_disable_fp8_mha`` attribute. - Fix ONNX FP16 conversion failing to preserve public output types when type inference changes a graph output declaration before output casts are inserted. diff --git a/modelopt/torch/export/plugins/vllm_fakequant_hf.py b/modelopt/torch/export/plugins/vllm_fakequant_hf.py index 94ea80032..47f7c4639 100644 --- a/modelopt/torch/export/plugins/vllm_fakequant_hf.py +++ b/modelopt/torch/export/plugins/vllm_fakequant_hf.py @@ -48,10 +48,16 @@ from modelopt.torch.quantization.utils.core_utils import ( ) from modelopt.torch.quantization.utils.layerwise_calib import LayerActivationCollector from modelopt.torch.utils import get_unwrapped_name, safe_save +from modelopt.torch.utils.plugins.hf_checkpoint_utils import copy_off_index_safetensors from ..layer_utils import get_experts_list from ..quant_utils import get_quantization_format -from ..unified_export_hf import collect_shared_input_modules, read_unplaced_weights +from ..unified_export_hf import ( + _copy_non_model_files_from_source, + _source_checkpoint, + collect_shared_input_modules, + read_unplaced_weights, +) __all__ = [ "export_hf_vllm_fq_checkpoint", @@ -755,6 +761,8 @@ def export_hf_vllm_fq_checkpoint( # inplace_mem_efficient branch (it deliberately omits state_dict= there -- see the # comment above -- so there is no state_dict to merge extras into). _carry_over_unplaced_weights(export_dir, model) + copy_off_index_safetensors(_source_checkpoint(model), export_dir) + _copy_non_model_files_from_source(model, export_dir) finally: if not inplace_mem_efficient: diff --git a/modelopt/torch/export/quant_utils.py b/modelopt/torch/export/quant_utils.py index 7d2d5d847..f80b552f9 100755 --- a/modelopt/torch/export/quant_utils.py +++ b/modelopt/torch/export/quant_utils.py @@ -1700,7 +1700,7 @@ def _get_carried_over_module_names(model: nn.Module) -> list[str]: module is invisible differs (no quantizer there, no module at all here). Prefers ``_modelopt_carried_over_names``, which the export records once it knows what it - actually wrote -- carried tensors plus the off-index sidecars copied verbatim. Those sidecars + actually wrote -- carried tensors plus the off-index weight files copied verbatim. Those files are never ``unexpected_keys``, so the unplaced list alone would miss GLM-4.7's ``mtp.safetensors`` and leave its tensors in the export with nothing in ``exclude_modules``. diff --git a/modelopt/torch/export/unified_export_hf.py b/modelopt/torch/export/unified_export_hf.py index 64e158c05..4551c57c1 100644 --- a/modelopt/torch/export/unified_export_hf.py +++ b/modelopt/torch/export/unified_export_hf.py @@ -20,7 +20,6 @@ import copy import importlib import json import re -import shutil import tempfile import warnings from builtins import ValueError @@ -37,6 +36,8 @@ from safetensors.torch import save_file from modelopt.torch.models import hf_model_type, is_moe from modelopt.torch.utils.plugins.hf_checkpoint_utils import ( + copy_non_model_files, + copy_off_index_safetensors, locate_source_keys, off_index_safetensors_files, sanitize_hf_config_for_deployment, @@ -1510,13 +1511,34 @@ def _sanitize_generation_config_for_save(model: torch.nn.Module) -> None: gc.do_sample = True +def _copy_non_model_files_from_source(model: nn.Module, export_dir: "str | Path") -> None: + """Copy the source checkpoint's non-model files into the export, from local disk only. + + See ``copy_non_model_files``. Exporters never fetch from the Hub: a model loaded by Hub ID + gets a warning instead, since its cache holds only the files loading needed. + """ + source = _source_checkpoint(model) + if source is None: + return + if not Path(source).is_dir(): + warnings.warn( + f"The source checkpoint {source!r} is not a local directory, so its non-model files " + "(tokenizer, processor, remote code, chat templates, ...) were not copied into the " + "export. Get a local copy first with modelopt.torch.export.ensure_local_checkpoint " + "and load the model from that directory." + ) + return + copy_non_model_files(source, export_dir) + + def save_non_weight_artifacts(model: nn.Module, export_dir: Path) -> None: - """Write config.json, generation_config.json, and trust_remote_code modeling files. + """Write config.json and generation_config.json, then copy the source's non-model files. For exporters that stream weights out themselves and never hand a state dict to ``save_pretrained``, which is not an option here: MoE models (e.g. DSR1) share expert storage across layers, so safetensors' shared-tensor check fires even on an empty dict. - The ``*.py`` files are what ``trust_remote_code`` checkpoints (e.g. NemotronH) need. + The non-model files include the ``*.py`` modules ``trust_remote_code`` checkpoints (e.g. + NemotronH) need. """ _sanitize_generation_config_for_save(model) # transformers' own revert_weight_conversion cannot handle quantized state dicts. @@ -1530,12 +1552,7 @@ def save_non_weight_artifacts(model: nn.Module, export_dir: Path) -> None: with contextlib.suppress(Exception): model.generation_config.save_pretrained(str(export_dir)) - src_dir = Path(getattr(model.config, "_name_or_path", "") or "") - if src_dir.is_dir(): - for py_file in src_dir.glob("*.py"): - dst = export_dir / py_file.name - if not dst.exists(): - shutil.copy2(py_file, dst) + _copy_non_model_files_from_source(model, export_dir) def export_speculative_decoding( @@ -1613,7 +1630,7 @@ def _source_checkpoint(model: nn.Module) -> str | None: Prefers ``_modelopt_source_checkpoint`` (recorded at load time by ``record_unplaced_source_keys``). A model that reached export without going through that path still knows its own provenance via ``config._name_or_path``, and the several places this is - asked must agree on the answer -- otherwise, for instance, a weight is carried but its sidecar + asked must agree on the answer -- otherwise, for instance, a weight is carried but its off-index tensors never reach ``exclude_modules`` because the two halves disagreed about where the checkpoint was. """ @@ -1648,15 +1665,17 @@ def carryable_unplaced_keys(model: nn.Module) -> list[str]: def off_index_tensor_names(model: nn.Module) -> list[str]: - """Tensor names in the checkpoint's off-index safetensors sidecars. + """Tensor names in the checkpoint's off-index weight files. - Those files (GLM-4.7's ``mtp.safetensors``) are copied into the export verbatim rather than - loaded, so they are never ``unexpected_keys`` and :func:`carryable_unplaced_keys` cannot see - them -- yet their tensors land in the export in original precision exactly like a carried - weight, and must reach ``exclude_modules`` the same way. Before this mechanism existed + Those files (GLM-4.7's ``mtp.safetensors``) are copied into the export verbatim by + :func:`~modelopt.torch.utils.plugins.hf_checkpoint_utils.copy_off_index_safetensors` rather + than loaded, so they are never ``unexpected_keys`` and + :func:`carryable_unplaced_keys` cannot see them -- yet their tensors land in the export in + original precision exactly like a carried weight, and must reach ``exclude_modules`` the same + way. Before this mechanism existed ``_add_mtp_exclusions`` covered them by globbing for ``mtp*``. - Reads safetensors headers only, never tensor data, and stays silent when the sidecars or the + Reads safetensors headers only, never tensor data, and stays silent when those files or the library cannot be read: an absent exclusion is a deployment problem, but so is an export that dies while computing one. """ @@ -1869,7 +1888,7 @@ def export_hf_checkpoint( if _writes_extra and _carried: extra_state_dict = {**_carried, **(extra_state_dict or {})} # Everything the export writes in original precision straight from the source, by either - # mechanism: tensors carried above, and the off-index sidecars copied verbatim alongside. + # mechanism: tensors carried above, and the off-index weight files copied verbatim alongside. # get_quant_config reads this to seed exclude_modules; recorded here because it runs before # that, and because only this point knows what was actually written rather than what was # merely unplaced. @@ -1881,6 +1900,10 @@ def export_hf_checkpoint( if exporter is not None: # Per-layer export wrote the shards during calibration; this writes the rest. exporter.finalize(extra_state_dict=extra_state_dict) + # Into the exporter's directory, which is where finalize() wrote: export_dir may differ + # when the recipe's layerwise.export_dir chose it. + if _writes_extra: + copy_off_index_safetensors(_source_checkpoint(model), exporter.export_dir) return export_dir = Path(export_dir) @@ -2014,6 +2037,10 @@ def export_hf_checkpoint( if rank == 0: _write_hf_export_config(model, hf_quant_config, export_dir) + copy_off_index_safetensors(_source_checkpoint(model), export_dir) + if not (_offloaded or is_fsdp2_sharded): + # The streaming paths already did, from save_non_weight_artifacts. + _copy_non_model_files_from_source(model, export_dir) except Exception as e: warnings.warn( diff --git a/tests/unit/torch/export/test_unified_export_hf.py b/tests/unit/torch/export/test_unified_export_hf.py index 4827762a8..9c9467ba1 100644 --- a/tests/unit/torch/export/test_unified_export_hf.py +++ b/tests/unit/torch/export/test_unified_export_hf.py @@ -779,6 +779,8 @@ def test_layerwise_finalize_sees_the_carried_keys(tmp_path, monkeypatch): seen = {} class _Exporter: + export_dir = tmp_path + def finalize(self, extra_state_dict=None): seen["keys"] = getattr(model, "_modelopt_carried_over_names", "") return {} @@ -797,6 +799,94 @@ def test_layerwise_finalize_sees_the_carried_keys(tmp_path, monkeypatch): ) +def test_export_writes_the_off_index_safetensors(tmp_path, monkeypatch): + """Off-index safetensors (GLM-4.7's mtp.safetensors) are weights, so the export writes them. + + Their tensors are already seeded into exclude_modules by off_index_tensor_names; if the files + themselves did not reach the export, the checkpoint would name exclusions for weights it lacks. + On the layerwise path they must land where finalize() wrote -- the exporter's own directory, + which a recipe's layerwise.export_dir can set apart from the export_dir argument. + """ + from modelopt.torch.export import unified_export_hf as uehf + from modelopt.torch.export.layerwise_export import LAYERWISE_EXPORTER_ATTR + + ckpt, export = tmp_path / "ckpt", tmp_path / "export" + ckpt.mkdir() + export.mkdir() # LayerwiseExporter creates it when it binds, before calibration + save_file({"a.weight": torch.zeros(2)}, str(ckpt / "model-00001-of-00001.safetensors")) + (ckpt / "model.safetensors.index.json").write_text( + '{"weight_map": {"a.weight": "model-00001-of-00001.safetensors"}}' + ) + save_file({"model.mtp.eh_proj.weight": torch.ones(2)}, str(ckpt / "mtp.safetensors")) + + class _Exporter: + export_dir = export + + def finalize(self, extra_state_dict=None): + return {} + + model = _ProvenanceModel(name_or_path=ckpt) + model._modelopt_source_checkpoint = str(ckpt) + setattr(model, LAYERWISE_EXPORTER_ATTR, _Exporter()) + monkeypatch.setattr(uehf, "read_unplaced_weights", lambda m, **kw: {}) + + uehf.export_hf_checkpoint(model, export_dir=tmp_path / "elsewhere") + + assert sorted(p.name for p in export.iterdir()) == ["mtp.safetensors"] + assert model._modelopt_carried_over_names == ["model.mtp.eh_proj.weight"] + + +def test_exporters_copy_non_model_files_from_a_local_source(tmp_path): + from modelopt.torch.export import unified_export_hf as uehf + + ckpt, export = tmp_path / "ckpt", tmp_path / "export" + ckpt.mkdir() + export.mkdir() + (ckpt / "tokenizer.json").write_text("{}") + (ckpt / "model.safetensors").write_text("weights") + + uehf._copy_non_model_files_from_source(_ProvenanceModel(name_or_path=ckpt), export) + + assert sorted(p.name for p in export.iterdir()) == ["tokenizer.json"] + + +def test_export_hf_checkpoint_copies_the_source_non_model_files(tmp_path): + """The regular (non-streaming, non-layerwise) path copies them after save_pretrained.""" + from _test_utils.torch.transformers_models import create_tiny_llama_dir + from transformers import AutoModelForCausalLM + + from modelopt.torch.export import export_hf_checkpoint + + source = create_tiny_llama_dir(tmp_path) + (source / "README.md").write_text("# source\n") + (source / "assets").mkdir() + (source / "assets" / "notes.txt").write_text("asset\n") + model = AutoModelForCausalLM.from_pretrained(source) + export = tmp_path / "export" + + export_hf_checkpoint(model, export_dir=export) + + assert (export / "README.md").read_text() == "# source\n" + assert (export / "assets" / "notes.txt").read_text() == "asset\n" + # The export's own config, not the source's. + assert (export / "config.json").read_text() != (source / "config.json").read_text() + + +def test_exporters_warn_rather_than_fetch_for_a_hub_source(tmp_path, monkeypatch): + """A model loaded by Hub ID has only what loading needed in its cache: warn, do not download.""" + from modelopt.torch.export import unified_export_hf as uehf + from modelopt.torch.utils.plugins import hf_checkpoint_utils + + def fail(*args, **kwargs): + raise AssertionError("an exporter must not download") + + monkeypatch.setattr(hf_checkpoint_utils, "snapshot_download", fail) + + with pytest.warns(UserWarning, match="ensure_local_checkpoint"): + uehf._copy_non_model_files_from_source(_ProvenanceModel(name_or_path="org/model"), tmp_path) + assert list(tmp_path.iterdir()) == [] + + def test_carries_a_key_the_index_does_not_list(tmp_path): """A tensor inside a main shard but absent from weight_map must still be carried.