mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
Keep fp8 kv none and use nvfp4 KV for the autoquant case
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Signed-off-by: Chenjie Luo <chenjiel@nvidia.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
c6e60a12a7
commit
a700005703
@@ -69,8 +69,9 @@ def test_ptq_whisper(command):
|
||||
PTQCommand(recipe="general/ptq/iq2_xxs", kv_cache_quant="none"),
|
||||
PTQCommand(recipe="general/ptq/iq2_xs", kv_cache_quant="none"),
|
||||
PTQCommand(recipe="general/ptq/iq2_s", kv_cache_quant="none"),
|
||||
# Default KV cache (fp8_cast); fp8 and nvfp4 are also deployed with TRT-LLM
|
||||
# fp8 and nvfp4 checkpoints are also deployed with TRT-LLM
|
||||
PTQCommand(quant="fp8", min_sm=89),
|
||||
PTQCommand(quant="fp8", kv_cache_quant="none", min_sm=89),
|
||||
PTQCommand(quant="nvfp4"),
|
||||
PTQCommand(quant="mxfp8", min_sm=100),
|
||||
# Calibrated KV cache
|
||||
@@ -78,7 +79,7 @@ def test_ptq_whisper(command):
|
||||
# AutoQuantize recipe; KV via --kv_cache_quant fallback
|
||||
PTQCommand(
|
||||
recipe="general/auto_quantize/nvfp4_fp8_at_5p4bits",
|
||||
kv_cache_quant="fp8",
|
||||
kv_cache_quant="nvfp4",
|
||||
calib_batch_size=4,
|
||||
),
|
||||
# multi_gpu
|
||||
|
||||
Reference in New Issue
Block a user