mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
## What does this PR do? **Type of change:** refactor / deprecation (examples) Follow-up to #1705 (which consolidated `examples/vlm_ptq` into `examples/llm_ptq`). Since that example now covers Hugging Face **LLM and VLM** PTQ, the `llm_ptq` name is a misnomer. This renames the directory to `examples/hf_ptq` and leaves a relative symlink `examples/llm_ptq → hf_ptq` so existing paths/commands keep working during a deprecation window. Requested by @kevalmorabia97 on #1705 (with the symlink-for-back-compat approach), targeted for the **same 0.46 release** as the consolidation. ### Changes - `git mv examples/llm_ptq → examples/hf_ptq` and `tests/examples/llm_ptq → tests/examples/hf_ptq` (the CI runner maps the matrix name to both `examples/<name>` and `tests/examples/<name>`). - Add a tracked back-compat symlink `examples/llm_ptq → hf_ptq`. - Update CI matrices and all repo **path references** (docs, READMEs, agent skills, launcher/debugger tools, tests) from `llm_ptq` to `hf_ptq`. - Keep Python identifiers / test-util module names (`run_llm_ptq_command`, `llm_ptq_utils`) — they name the LLM-PTQ task, not the directory. - Preserve the CODEOWNERS team slug (`modelopt-examples-llm_ptq-codeowners`) and historical CHANGELOG entries; add a CHANGELOG deprecation note. ### Back-compat caveats (inherent to git directory symlinks) - ✅ Linux/macOS CLI usage and Python `cwd`/pytest resolution work through the symlink. - ⚠️ Windows git checkouts don't materialize symlinks by default (low impact — this example is Linux-only in practice). - ⚠️ GitHub web doesn't follow directory symlinks, so legacy external deep-links to `examples/llm_ptq/...` won't navigate in. All **internal** references are repointed to `hf_ptq`, so the symlink is only for legacy external/CLI use. ### Usage (unchanged via symlink) ```bash # New canonical path cd examples/hf_ptq scripts/huggingface_example.sh --model <hf_model> --quant fp8 # Old path still works (forwards via symlink) cd examples/llm_ptq && scripts/huggingface_example.sh --model <hf_model> --quant fp8 ``` ### Testing - `bash -n` on moved/edited shell scripts (new path + via symlink). - `py_compile` on moved/edited Python; test re-export shim repointed to `examples/hf_ptq/example_utils`. - Verified git tracks `examples/llm_ptq` as a single symlink (mode 120000), not a duplicated tree (no pre-commit / pytest double-processing). - `pre-commit run` on all changed files passes. ### Before your PR is "*Ready for review*" - Is this change backward compatible?: ✅ (relative symlink keeps `examples/llm_ptq` paths valid; see caveats above) - Did you write any new necessary tests?: N/A (pure rename; existing tests moved with the dir) - Did you update Changelog?: ✅ ### Additional Information Follow-up (later release): remove the `examples/llm_ptq` symlink once external references have migrated. 🤖 Generated with [Claude Code](https://claude.com/claude-code) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * PTQ guidance now directs to the unified Hugging Face PTQ flow, including VLM quantization via the shared `--vlm` entry point. * **Documentation** * Updated README and guide links, references, and command snippets to use `hf_ptq` (replacing `llm_ptq`). * Deprecated and consolidated `vlm_ptq` into `hf_ptq`; removed VILA/NVILA coverage from the Hugging Face PTQ examples. * **Bug Fixes** * Improved detection and routing so local/manual setup uses the correct PTQ source. * **Tests / Chores** * CI and example tests updated to run the `hf_ptq` variants. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Zhiyu Cheng <zhiyuc@nvidia.com> Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
146 lines
5.5 KiB
Python
146 lines
5.5 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""End-to-end CI test for the offline speculative decoding PTQ workflow.
|
|
|
|
Covers the three-stage pipeline:
|
|
1. Collect hidden states from the base model → .pt files
|
|
2. Train an offline EAGLE draft model → ModelOpt checkpoint
|
|
3. PTQ the offline checkpoint → quantized export
|
|
|
|
Running all three stages in sequence validates that the data format produced
|
|
by stage 1 is correctly consumed by stage 2 and that the checkpoint produced
|
|
by stage 2 is correctly quantized in stage 3.
|
|
"""
|
|
|
|
import pytest
|
|
import safetensors.torch
|
|
import torch
|
|
from _test_utils.examples.run_command import MODELOPT_ROOT, run_example_command
|
|
|
|
from modelopt.torch.export.plugins.hf_spec_export import LLAMA_EAGLE_SINGLE_LAYER
|
|
|
|
EAGLE3_YAML = str(
|
|
MODELOPT_ROOT / "modelopt_recipes" / "general" / "speculative_decoding" / "eagle3.yaml"
|
|
)
|
|
|
|
# Tiny EAGLE architecture overrides (dotlist entries)
|
|
_TINY_EAGLE_ARCH = [
|
|
"eagle.eagle_architecture_config.max_position_embeddings=128",
|
|
"eagle.eagle_architecture_config.num_hidden_layers=1",
|
|
"eagle.eagle_architecture_config.intermediate_size=64",
|
|
"eagle.eagle_architecture_config.num_attention_heads=2",
|
|
"eagle.eagle_architecture_config.num_key_value_heads=2",
|
|
"eagle.eagle_architecture_config.head_dim=64",
|
|
]
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def offline_ptq_dirs(tmp_path_factory):
|
|
"""Shared output directories for all stages."""
|
|
return {
|
|
"hidden_states": tmp_path_factory.mktemp("hidden_states"),
|
|
"eagle_ckpt": tmp_path_factory.mktemp("eagle_ckpt"),
|
|
"ptq_export": tmp_path_factory.mktemp("ptq_export"),
|
|
}
|
|
|
|
|
|
def test_collect_hidden_states(tiny_llama_path, tiny_conversations_path, offline_ptq_dirs):
|
|
"""Stage 1: generate .pt hidden state files from the base model."""
|
|
run_example_command(
|
|
[
|
|
"python",
|
|
"collect_hidden_states/compute_hidden_states_hf.py",
|
|
"--model",
|
|
tiny_llama_path,
|
|
"--input-data",
|
|
str(tiny_conversations_path),
|
|
"--output-dir",
|
|
str(offline_ptq_dirs["hidden_states"]),
|
|
"--debug-max-num-conversations",
|
|
"2",
|
|
"--max-seq-len",
|
|
"32",
|
|
],
|
|
"speculative_decoding",
|
|
)
|
|
|
|
pt_files = list(offline_ptq_dirs["hidden_states"].glob("*.pt"))
|
|
assert len(pt_files) > 0, "No .pt files generated by compute_hidden_states_hf.py"
|
|
|
|
# Validate the format expected by OfflineSupervisedDataset
|
|
sample = torch.load(str(pt_files[0]))
|
|
assert "input_ids" in sample, "Missing 'input_ids' in .pt file"
|
|
assert "hidden_states" in sample, "Missing 'hidden_states' in .pt file"
|
|
|
|
|
|
def test_offline_eagle_training(tiny_llama_path, tiny_daring_anteater_path, offline_ptq_dirs):
|
|
"""Stage 2: train an EAGLE3 draft model using the offline hidden states."""
|
|
output_dir = offline_ptq_dirs["eagle_ckpt"] / "trained"
|
|
|
|
overrides = [
|
|
f"model.model_name_or_path={tiny_llama_path}",
|
|
f"data.data_path={tiny_daring_anteater_path}",
|
|
f"data.offline_data_path={offline_ptq_dirs['hidden_states']}",
|
|
f"training.output_dir={output_dir}",
|
|
"training.num_train_epochs=1",
|
|
"training.learning_rate=1e-5",
|
|
"training.training_seq_len=64",
|
|
"training.save_steps=1",
|
|
# torch.compile is smoke-tested once by test_llama_eagle3[1-False]; skip its warmup here.
|
|
"eagle.eagle_use_torch_compile=false",
|
|
*_TINY_EAGLE_ARCH,
|
|
]
|
|
|
|
run_example_command(
|
|
["./launch_train.sh", "--config", EAGLE3_YAML, *overrides],
|
|
"speculative_decoding",
|
|
setup_free_port=True,
|
|
)
|
|
|
|
assert output_dir.exists(), "EAGLE training did not produce an output directory"
|
|
|
|
|
|
def test_offline_ptq(offline_ptq_dirs):
|
|
"""Stage 3: run PTQ on the offline EAGLE checkpoint using the hidden state dataset."""
|
|
run_example_command(
|
|
[
|
|
"python",
|
|
"hf_ptq.py",
|
|
"--pyt_ckpt_path",
|
|
str(offline_ptq_dirs["eagle_ckpt"] / "trained"),
|
|
"--qformat",
|
|
"fp8",
|
|
"--calib_size",
|
|
"2",
|
|
"--batch_size",
|
|
"1",
|
|
"--specdec_offline_dataset",
|
|
str(offline_ptq_dirs["hidden_states"]),
|
|
"--export_path",
|
|
str(offline_ptq_dirs["ptq_export"]),
|
|
],
|
|
"hf_ptq",
|
|
)
|
|
|
|
# Verify the exported checkpoint exists and has the expected EAGLE keys
|
|
export_dir = offline_ptq_dirs["ptq_export"]
|
|
assert (export_dir / "model.safetensors").exists(), "PTQ export missing model.safetensors"
|
|
assert (export_dir / "config.json").exists(), "PTQ export missing config.json"
|
|
|
|
state_dict = safetensors.torch.load_file(export_dir / "model.safetensors")
|
|
for key in LLAMA_EAGLE_SINGLE_LAYER["required"] - {"fc", "layers.0.hidden_norm"}:
|
|
assert f"{key}.weight" in state_dict, f"Missing key '{key}.weight' in exported state dict"
|