mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? Type of change: new example <!-- Use one of the following: Bug fix, new feature, new example, new tests, documentation. --> Add deepseek v4 official modeling ptq example ### Usage See readme, and it requires the vllm PR: https://github.com/vllm-project/vllm/pull/42209 ### Testing Tested with ptq and export of dsv4 flash and served with vllm. ### Before your PR is "*Ready for review*" Make sure you read and follow [Contributor guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md) and your commits are signed (`git commit -s -S`). Make sure you read and follow the [Security Best Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors) (e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(..., weights_only=False)`, `pickle`, etc.). - Is this change backward compatible?: ✅ / ❌ / N/A <!--- If ❌, explain why. --> - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: ✅ / ❌ / N/A <!--- Mandatory --> - Did you write any new necessary tests?: ✅ / ❌ / N/A <!--- Mandatory for new features or examples. --> - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: ✅ / ❌ / N/A <!--- Only for new features, API changes, critical bug fixes or backward incompatible changes. --> ### Additional Information <!-- E.g. related issue. --> <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added DeepSeek‑V4 routed‑expert post‑training quantization and an NVFP4 checkpoint conversion utility. * **Documentation** * Expanded DeepSeek quantization guide with directory layout, updated V3/V3.2 workflows, and detailed V4 routed‑expert calibration, single/multi‑node examples, and export guidance. * **Chores** * Made example quantization scripts location‑independent. * Updated pre‑commit license hook to skip DeepSeek example quantization files. <!-- review_stack_entry_start --> [](https://app.coderabbit.ai/change-stack/NVIDIA/Model-Optimizer/pull/1341?utm_source=github_walkthrough&utm_medium=github&utm_campaign=change_stack) <!-- review_stack_entry_end --> <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Meng Xin <mxin@nvidia.com>
178 lines
6.4 KiB
YAML
178 lines
6.4 KiB
YAML
# NOTE: Make sure to update version in dev requirements (pyproject.toml) as well!
|
|
repos:
|
|
- repo: https://github.com/pre-commit/pre-commit-hooks
|
|
rev: v6.0.0
|
|
hooks:
|
|
- id: check-added-large-files
|
|
args: [--maxkb=500, --enforce-all]
|
|
exclude: >
|
|
(?x)^(
|
|
uv.lock|
|
|
examples/diffusers/quantization/assets/.*.png|
|
|
examples/diffusers/cache_diffusion/assets/.*.png|
|
|
)$
|
|
- id: check-json
|
|
exclude: ^.vscode/.*.json # vscode files can take comments
|
|
- id: check-merge-conflict
|
|
- id: check-symlinks
|
|
- id: check-toml
|
|
- id: mixed-line-ending
|
|
args: [--fix=lf]
|
|
- id: requirements-txt-fixer
|
|
|
|
- repo: https://github.com/astral-sh/ruff-pre-commit
|
|
rev: v0.12.11
|
|
hooks:
|
|
- id: ruff-check
|
|
args: [--fix, --exit-non-zero-on-fix]
|
|
exclude: ^examples/specdec_bench/specdec_bench/datasets/speed\.py$
|
|
- id: ruff-format
|
|
exclude: ^examples/specdec_bench/specdec_bench/datasets/speed\.py$
|
|
|
|
- repo: https://github.com/pre-commit/mirrors-mypy
|
|
rev: v1.17.1
|
|
hooks:
|
|
- id: mypy
|
|
|
|
- repo: https://github.com/pre-commit/mirrors-clang-format
|
|
rev: v21.1.0
|
|
hooks:
|
|
- id: clang-format
|
|
types_or: [c++, c, c#, cuda, java, javascript, objective-c, proto] # no json!
|
|
args: ["--style={ColumnLimit: 100}"]
|
|
|
|
- repo: https://github.com/pre-commit/pygrep-hooks
|
|
rev: v1.10.0
|
|
hooks:
|
|
- id: rst-backticks
|
|
- id: rst-directive-colons
|
|
- id: rst-inline-touching-normal
|
|
|
|
- repo: https://github.com/jumanjihouse/pre-commit-hook-yamlfmt
|
|
rev: 0.2.3
|
|
hooks:
|
|
- id: yamlfmt
|
|
args: [--mapping=2, --sequence=4, --offset=2, --implicit_start, --implicit_end, --preserve-quotes]
|
|
exclude: ^.github/workflows/
|
|
|
|
- repo: local
|
|
hooks:
|
|
- id: normalize-yaml-ext
|
|
name: normalize .yml to .yaml in required places, right now only yaml files in modelopt_recipes
|
|
entry: python tools/precommit/normalize_yaml_ext.py
|
|
language: system
|
|
files: ^modelopt_recipes/.*\.yml$
|
|
|
|
- id: check-modelopt-recipes
|
|
name: validate modelopt recipes
|
|
entry: python tools/precommit/check_modelopt_recipes.py
|
|
language: system
|
|
files: ^modelopt_recipes/
|
|
# configs/ contains reusable snippets (not full recipes) — skip recipe validation
|
|
exclude: ^modelopt_recipes/configs/
|
|
|
|
# Instructions to change license file if ever needed:
|
|
# https://github.com/Lucas-C/pre-commit-hooks#removing-old-license-and-replacing-it-with-a-new-one
|
|
- repo: https://github.com/Lucas-C/pre-commit-hooks
|
|
rev: v1.5.5
|
|
hooks:
|
|
# Default hook for Apache 2.0 in python and shell files
|
|
- id: insert-license
|
|
alias: insert-license-py
|
|
args:
|
|
- --license-filepath
|
|
- ./LICENSE_HEADER
|
|
- --comment-style
|
|
- "#"
|
|
- --allow-past-years
|
|
types_or: [python, shell]
|
|
# NOTE: Exclude files that have copyright or license headers from another company or individual
|
|
# since we want to keep those above the license header added by this hook.
|
|
# Instead, we should manually add the license header to those files *after* the original header.
|
|
exclude: >
|
|
(?x)^(
|
|
modelopt/torch/quantization/utils/calib_utils.py|
|
|
modelopt/onnx/quantization/operators.py|
|
|
modelopt/onnx/quantization/ort_patching.py|
|
|
modelopt/torch/_deploy/utils/onnx_utils.py|
|
|
modelopt/torch/export/transformer_engine.py|
|
|
modelopt/torch/puzzletron/anymodel/models/gpt_oss/gpt_oss_pruned_to_mxfp4.py|
|
|
modelopt/torch/quantization/export_onnx.py|
|
|
modelopt/torch/quantization/plugins/attention.py|
|
|
modelopt/torch/sparsity/attention_sparsity/methods/vsa_utils.py|
|
|
modelopt/torch/speculative/eagle/utils.py|
|
|
modelopt/torch/speculative/plugins/hf_medusa.py|
|
|
modelopt/torch/utils/plugins/megatron_mmlu.py|
|
|
examples/deepseek/deepseek_v3/quantize_to_nvfp4.py|
|
|
examples/deepseek/deepseek_v3/ptq.py|
|
|
examples/diffusers/quantization/onnx_utils/export.py|
|
|
examples/llm_eval/lm_eval_hf.py|
|
|
examples/llm_eval/mmlu.py|
|
|
examples/llm_eval/modeling.py|
|
|
examples/llm_qat/train.py|
|
|
examples/llm_sparsity/weight_sparsity/finetune.py|
|
|
examples/specdec_bench/specdec_bench/models/specbench_medusa.py|
|
|
examples/speculative_decoding/main.py|
|
|
examples/speculative_decoding/medusa_utils.py|
|
|
examples/speculative_decoding/scripts/server_generate.py|
|
|
experimental/dms/models/qwen3/configuration_qwen3_dms.py|
|
|
experimental/dms/models/qwen3/modeling_qwen3_dms.py|
|
|
)$
|
|
|
|
# Default hook for Apache 2.0 in c/c++/cuda files
|
|
- id: insert-license
|
|
alias: insert-license-c
|
|
args:
|
|
- --license-filepath
|
|
- ./LICENSE_HEADER
|
|
- --comment-style
|
|
- "/*| *| */"
|
|
- --allow-past-years
|
|
types_or: [c++, cuda, c]
|
|
|
|
- repo: https://github.com/PyCQA/bandit
|
|
rev: 1.7.9
|
|
hooks:
|
|
- id: bandit
|
|
args: ["-c", "pyproject.toml", "-q"]
|
|
additional_dependencies: ["bandit[toml]"]
|
|
|
|
- repo: local
|
|
hooks:
|
|
- id: generate-arguments-md
|
|
name: Regenerate examples/llm_qat/ARGUMENTS.md
|
|
entry: bash -c 'python examples/llm_qat/arguments.py --generate_docs examples/llm_qat/ARGUMENTS.md'
|
|
language: system
|
|
files: >-
|
|
(?x)^(
|
|
examples/llm_qat/arguments\.py|
|
|
modelopt/torch/distill/plugins/huggingface\.py|
|
|
modelopt/torch/opt/plugins/transformers\.py|
|
|
modelopt/torch/quantization/plugins/transformers_trainer\.py
|
|
)$
|
|
pass_filenames: false
|
|
|
|
- repo: https://github.com/DavidAnson/markdownlint-cli2
|
|
rev: v0.18.1
|
|
hooks:
|
|
- id: markdownlint-cli2
|
|
args: ["--fix"]
|
|
|
|
##### Manual hooks (Expect many false positives)
|
|
# These hooks are only run with `pre-commit run --all-files --hook-stage manual <hook_id>`
|
|
|
|
# Spell checker
|
|
- repo: https://github.com/crate-ci/typos
|
|
rev: v1.35.8
|
|
hooks:
|
|
- id: typos
|
|
stages: [manual]
|
|
|
|
# Link checker
|
|
- repo: https://github.com/lycheeverse/lychee.git
|
|
rev: v0.15.1
|
|
hooks:
|
|
- id: lychee
|
|
args: ["--no-progress", "--exclude-loopback"]
|
|
stages: [manual]
|