HF exporters write off-index safetensors and non-model files

export_hf_checkpoint seeded the tensors of off-index safetensors (GLM-4.7's
mtp.safetensors) into exclude_modules but left copying the files to the
hf_ptq example, so any other caller exported exclusions for weights the
checkpoint lacked. It likewise left the source's non-model files --
tokenizer, processor, remote code, chat templates -- to the caller, while
its streaming paths copied only top-level *.py.

Every HF export path now does both itself, from the model's source
checkpoint on local disk: the regular, streaming and FSDP2 paths, the
layerwise path (into the exporter's own directory, which a recipe's
layerwise.export_dir can set apart from export_dir), and the vLLM
fake-quant export. save_non_weight_artifacts copies the non-model files in
place of its *.py glob. A source that is a Hub ID gets a warning instead of
a fetch: its cache holds only what loading needed.

Signed-off-by: Shengliang Xu <shengliangx@nvidia.com>
This commit is contained in:
Shengliang Xu
2026-10-01 18:25:46 +00:00
parent 2d71a30ff7
commit 137a128a8c
5 changed files with 146 additions and 19 deletions
+2
View File
@@ -40,6 +40,7 @@ Changelog
*Misc*
- A tracked ``examples/hf_ptq/hf_ptq.py`` run now writes ``.experiment.json`` into ``--export_path`` and uploads the same file with the run, so a checkpoint on disk names the experiment and MLflow run id that produced it. The pointer is written only once the export completes, and an export that is not tracked removes one it would otherwise inherit from a reused ``--export_path`` or from a quantized source checkpoint.
- ``export_hf_checkpoint`` and the vLLM fake-quant export now carry every non-model file of the source checkpoint -- everything but the weights and the metadata the export writes -- into the export verbatim, subdirectories included, instead of regenerating tokenizer and processor files. They read the source from local disk only, so get a local copy of a Hugging Face Hub model first with ``modelopt.torch.export.ensure_local_checkpoint`` -- which, under ``torch.distributed``, downloads once on rank 0 of the default group or of the ``group`` passed -- and load it from that directory.
**Backward Breaking Changes**
@@ -90,6 +91,7 @@ Changelog
**Bug Fixes**
- Fix Hugging Face exports of checkpoints with off-index safetensors (such as GLM-4.7's ``mtp.safetensors``) missing those files: ``export_hf_checkpoint`` now writes them itself.
- Fix Megatron-Core checkpoint saving for quantized grouped MoE experts when tensor and expert parallelism are both enabled.
- Fix shared ONNX export metadata and Diffusers attention policy: every ``NVFP4QuantExporter`` post-process now upgrades the default-domain opset to at least 23, all FP8 custom-op exports re-run ONNX shape/type inference after setting output metadata, and quantized SDPA derives FP8 MHA enablement from the live Q/K/V quantizers instead of honoring a caller-set ``_disable_fp8_mha`` attribute.
- Fix ONNX FP16 conversion failing to preserve public output types when type inference changes a graph output declaration before output casts are inserted.
@@ -48,10 +48,16 @@ from modelopt.torch.quantization.utils.core_utils import (
)
from modelopt.torch.quantization.utils.layerwise_calib import LayerActivationCollector
from modelopt.torch.utils import get_unwrapped_name, safe_save
from modelopt.torch.utils.plugins.hf_checkpoint_utils import copy_off_index_safetensors
from ..layer_utils import get_experts_list
from ..quant_utils import get_quantization_format
from ..unified_export_hf import collect_shared_input_modules, read_unplaced_weights
from ..unified_export_hf import (
_copy_non_model_files_from_source,
_source_checkpoint,
collect_shared_input_modules,
read_unplaced_weights,
)
__all__ = [
"export_hf_vllm_fq_checkpoint",
@@ -755,6 +761,8 @@ def export_hf_vllm_fq_checkpoint(
# inplace_mem_efficient branch (it deliberately omits state_dict= there -- see the
# comment above -- so there is no state_dict to merge extras into).
_carry_over_unplaced_weights(export_dir, model)
copy_off_index_safetensors(_source_checkpoint(model), export_dir)
_copy_non_model_files_from_source(model, export_dir)
finally:
if not inplace_mem_efficient:
+1 -1
View File
@@ -1700,7 +1700,7 @@ def _get_carried_over_module_names(model: nn.Module) -> list[str]:
module is invisible differs (no quantizer there, no module at all here).
Prefers ``_modelopt_carried_over_names``, which the export records once it knows what it
actually wrote -- carried tensors plus the off-index sidecars copied verbatim. Those sidecars
actually wrote -- carried tensors plus the off-index weight files copied verbatim. Those files
are never ``unexpected_keys``, so the unplaced list alone would miss GLM-4.7's
``mtp.safetensors`` and leave its tensors in the export with nothing in ``exclude_modules``.
+44 -17
View File
@@ -20,7 +20,6 @@ import copy
import importlib
import json
import re
import shutil
import tempfile
import warnings
from builtins import ValueError
@@ -37,6 +36,8 @@ from safetensors.torch import save_file
from modelopt.torch.models import hf_model_type, is_moe
from modelopt.torch.utils.plugins.hf_checkpoint_utils import (
copy_non_model_files,
copy_off_index_safetensors,
locate_source_keys,
off_index_safetensors_files,
sanitize_hf_config_for_deployment,
@@ -1510,13 +1511,34 @@ def _sanitize_generation_config_for_save(model: torch.nn.Module) -> None:
gc.do_sample = True
def _copy_non_model_files_from_source(model: nn.Module, export_dir: "str | Path") -> None:
"""Copy the source checkpoint's non-model files into the export, from local disk only.
See ``copy_non_model_files``. Exporters never fetch from the Hub: a model loaded by Hub ID
gets a warning instead, since its cache holds only the files loading needed.
"""
source = _source_checkpoint(model)
if source is None:
return
if not Path(source).is_dir():
warnings.warn(
f"The source checkpoint {source!r} is not a local directory, so its non-model files "
"(tokenizer, processor, remote code, chat templates, ...) were not copied into the "
"export. Get a local copy first with modelopt.torch.export.ensure_local_checkpoint "
"and load the model from that directory."
)
return
copy_non_model_files(source, export_dir)
def save_non_weight_artifacts(model: nn.Module, export_dir: Path) -> None:
"""Write config.json, generation_config.json, and trust_remote_code modeling files.
"""Write config.json and generation_config.json, then copy the source's non-model files.
For exporters that stream weights out themselves and never hand a state dict to
``save_pretrained``, which is not an option here: MoE models (e.g. DSR1) share expert
storage across layers, so safetensors' shared-tensor check fires even on an empty dict.
The ``*.py`` files are what ``trust_remote_code`` checkpoints (e.g. NemotronH) need.
The non-model files include the ``*.py`` modules ``trust_remote_code`` checkpoints (e.g.
NemotronH) need.
"""
_sanitize_generation_config_for_save(model)
# transformers' own revert_weight_conversion cannot handle quantized state dicts.
@@ -1530,12 +1552,7 @@ def save_non_weight_artifacts(model: nn.Module, export_dir: Path) -> None:
with contextlib.suppress(Exception):
model.generation_config.save_pretrained(str(export_dir))
src_dir = Path(getattr(model.config, "_name_or_path", "") or "")
if src_dir.is_dir():
for py_file in src_dir.glob("*.py"):
dst = export_dir / py_file.name
if not dst.exists():
shutil.copy2(py_file, dst)
_copy_non_model_files_from_source(model, export_dir)
def export_speculative_decoding(
@@ -1613,7 +1630,7 @@ def _source_checkpoint(model: nn.Module) -> str | None:
Prefers ``_modelopt_source_checkpoint`` (recorded at load time by
``record_unplaced_source_keys``). A model that reached export without going through that path
still knows its own provenance via ``config._name_or_path``, and the several places this is
asked must agree on the answer -- otherwise, for instance, a weight is carried but its sidecar
asked must agree on the answer -- otherwise, for instance, a weight is carried but its off-index
tensors never reach ``exclude_modules`` because the two halves disagreed about where the
checkpoint was.
"""
@@ -1648,15 +1665,17 @@ def carryable_unplaced_keys(model: nn.Module) -> list[str]:
def off_index_tensor_names(model: nn.Module) -> list[str]:
"""Tensor names in the checkpoint's off-index safetensors sidecars.
"""Tensor names in the checkpoint's off-index weight files.
Those files (GLM-4.7's ``mtp.safetensors``) are copied into the export verbatim rather than
loaded, so they are never ``unexpected_keys`` and :func:`carryable_unplaced_keys` cannot see
them -- yet their tensors land in the export in original precision exactly like a carried
weight, and must reach ``exclude_modules`` the same way. Before this mechanism existed
Those files (GLM-4.7's ``mtp.safetensors``) are copied into the export verbatim by
:func:`~modelopt.torch.utils.plugins.hf_checkpoint_utils.copy_off_index_safetensors` rather
than loaded, so they are never ``unexpected_keys`` and
:func:`carryable_unplaced_keys` cannot see them -- yet their tensors land in the export in
original precision exactly like a carried weight, and must reach ``exclude_modules`` the same
way. Before this mechanism existed
``_add_mtp_exclusions`` covered them by globbing for ``mtp*``.
Reads safetensors headers only, never tensor data, and stays silent when the sidecars or the
Reads safetensors headers only, never tensor data, and stays silent when those files or the
library cannot be read: an absent exclusion is a deployment problem, but so is an export that
dies while computing one.
"""
@@ -1869,7 +1888,7 @@ def export_hf_checkpoint(
if _writes_extra and _carried:
extra_state_dict = {**_carried, **(extra_state_dict or {})}
# Everything the export writes in original precision straight from the source, by either
# mechanism: tensors carried above, and the off-index sidecars copied verbatim alongside.
# mechanism: tensors carried above, and the off-index weight files copied verbatim alongside.
# get_quant_config reads this to seed exclude_modules; recorded here because it runs before
# that, and because only this point knows what was actually written rather than what was
# merely unplaced.
@@ -1881,6 +1900,10 @@ def export_hf_checkpoint(
if exporter is not None:
# Per-layer export wrote the shards during calibration; this writes the rest.
exporter.finalize(extra_state_dict=extra_state_dict)
# Into the exporter's directory, which is where finalize() wrote: export_dir may differ
# when the recipe's layerwise.export_dir chose it.
if _writes_extra:
copy_off_index_safetensors(_source_checkpoint(model), exporter.export_dir)
return
export_dir = Path(export_dir)
@@ -2014,6 +2037,10 @@ def export_hf_checkpoint(
if rank == 0:
_write_hf_export_config(model, hf_quant_config, export_dir)
copy_off_index_safetensors(_source_checkpoint(model), export_dir)
if not (_offloaded or is_fsdp2_sharded):
# The streaming paths already did, from save_non_weight_artifacts.
_copy_non_model_files_from_source(model, export_dir)
except Exception as e:
warnings.warn(
@@ -779,6 +779,8 @@ def test_layerwise_finalize_sees_the_carried_keys(tmp_path, monkeypatch):
seen = {}
class _Exporter:
export_dir = tmp_path
def finalize(self, extra_state_dict=None):
seen["keys"] = getattr(model, "_modelopt_carried_over_names", "<unset>")
return {}
@@ -797,6 +799,94 @@ def test_layerwise_finalize_sees_the_carried_keys(tmp_path, monkeypatch):
)
def test_export_writes_the_off_index_safetensors(tmp_path, monkeypatch):
"""Off-index safetensors (GLM-4.7's mtp.safetensors) are weights, so the export writes them.
Their tensors are already seeded into exclude_modules by off_index_tensor_names; if the files
themselves did not reach the export, the checkpoint would name exclusions for weights it lacks.
On the layerwise path they must land where finalize() wrote -- the exporter's own directory,
which a recipe's layerwise.export_dir can set apart from the export_dir argument.
"""
from modelopt.torch.export import unified_export_hf as uehf
from modelopt.torch.export.layerwise_export import LAYERWISE_EXPORTER_ATTR
ckpt, export = tmp_path / "ckpt", tmp_path / "export"
ckpt.mkdir()
export.mkdir() # LayerwiseExporter creates it when it binds, before calibration
save_file({"a.weight": torch.zeros(2)}, str(ckpt / "model-00001-of-00001.safetensors"))
(ckpt / "model.safetensors.index.json").write_text(
'{"weight_map": {"a.weight": "model-00001-of-00001.safetensors"}}'
)
save_file({"model.mtp.eh_proj.weight": torch.ones(2)}, str(ckpt / "mtp.safetensors"))
class _Exporter:
export_dir = export
def finalize(self, extra_state_dict=None):
return {}
model = _ProvenanceModel(name_or_path=ckpt)
model._modelopt_source_checkpoint = str(ckpt)
setattr(model, LAYERWISE_EXPORTER_ATTR, _Exporter())
monkeypatch.setattr(uehf, "read_unplaced_weights", lambda m, **kw: {})
uehf.export_hf_checkpoint(model, export_dir=tmp_path / "elsewhere")
assert sorted(p.name for p in export.iterdir()) == ["mtp.safetensors"]
assert model._modelopt_carried_over_names == ["model.mtp.eh_proj.weight"]
def test_exporters_copy_non_model_files_from_a_local_source(tmp_path):
from modelopt.torch.export import unified_export_hf as uehf
ckpt, export = tmp_path / "ckpt", tmp_path / "export"
ckpt.mkdir()
export.mkdir()
(ckpt / "tokenizer.json").write_text("{}")
(ckpt / "model.safetensors").write_text("weights")
uehf._copy_non_model_files_from_source(_ProvenanceModel(name_or_path=ckpt), export)
assert sorted(p.name for p in export.iterdir()) == ["tokenizer.json"]
def test_export_hf_checkpoint_copies_the_source_non_model_files(tmp_path):
"""The regular (non-streaming, non-layerwise) path copies them after save_pretrained."""
from _test_utils.torch.transformers_models import create_tiny_llama_dir
from transformers import AutoModelForCausalLM
from modelopt.torch.export import export_hf_checkpoint
source = create_tiny_llama_dir(tmp_path)
(source / "README.md").write_text("# source\n")
(source / "assets").mkdir()
(source / "assets" / "notes.txt").write_text("asset\n")
model = AutoModelForCausalLM.from_pretrained(source)
export = tmp_path / "export"
export_hf_checkpoint(model, export_dir=export)
assert (export / "README.md").read_text() == "# source\n"
assert (export / "assets" / "notes.txt").read_text() == "asset\n"
# The export's own config, not the source's.
assert (export / "config.json").read_text() != (source / "config.json").read_text()
def test_exporters_warn_rather_than_fetch_for_a_hub_source(tmp_path, monkeypatch):
"""A model loaded by Hub ID has only what loading needed in its cache: warn, do not download."""
from modelopt.torch.export import unified_export_hf as uehf
from modelopt.torch.utils.plugins import hf_checkpoint_utils
def fail(*args, **kwargs):
raise AssertionError("an exporter must not download")
monkeypatch.setattr(hf_checkpoint_utils, "snapshot_download", fail)
with pytest.warns(UserWarning, match="ensure_local_checkpoint"):
uehf._copy_non_model_files_from_source(_ProvenanceModel(name_or_path="org/model"), tmp_path)
assert list(tmp_path.iterdir()) == []
def test_carries_a_key_the_index_does_not_list(tmp_path):
"""A tensor inside a main shard but absent from weight_map must still be carried.