Keep fp8 kv none and use nvfp4 KV for the autoquant case

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Signed-off-by: Chenjie Luo <chenjiel@nvidia.com>
This commit is contained in:
Chenjie Luo
2026-09-29 23:06:54 +00:00
co-authored by Claude Opus 5.5
parent c6e60a12a7
commit a700005703
+3 -2
View File
@@ -69,8 +69,9 @@ def test_ptq_whisper(command):
PTQCommand(recipe="general/ptq/iq2_xxs", kv_cache_quant="none"),
PTQCommand(recipe="general/ptq/iq2_xs", kv_cache_quant="none"),
PTQCommand(recipe="general/ptq/iq2_s", kv_cache_quant="none"),
# Default KV cache (fp8_cast); fp8 and nvfp4 are also deployed with TRT-LLM
# fp8 and nvfp4 checkpoints are also deployed with TRT-LLM
PTQCommand(quant="fp8", min_sm=89),
PTQCommand(quant="fp8", kv_cache_quant="none", min_sm=89),
PTQCommand(quant="nvfp4"),
PTQCommand(quant="mxfp8", min_sm=100),
# Calibrated KV cache
@@ -78,7 +79,7 @@ def test_ptq_whisper(command):
# AutoQuantize recipe; KV via --kv_cache_quant fallback
PTQCommand(
recipe="general/auto_quantize/nvfp4_fp8_at_5p4bits",
kv_cache_quant="fp8",
kv_cache_quant="nvfp4",
calib_batch_size=4,
),
# multi_gpu