From 17439e653df905aac91b7aa208543223f07dc7ec Mon Sep 17 00:00:00 2001 From: Chenjie Luo <108829653+cjluo-nv@users.noreply.github.com> Date: Tue, 30 Sep 2025 11:24:35 -0700 Subject: [PATCH] Move phi4_mm warning to above (#389) Signed-off-by: Chenjie Luo <108829653+cjluo-nv@users.noreply.github.com> --- examples/llm_ptq/hf_ptq.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/examples/llm_ptq/hf_ptq.py b/examples/llm_ptq/hf_ptq.py index 0ac11f2f5..da6761252 100755 --- a/examples/llm_ptq/hf_ptq.py +++ b/examples/llm_ptq/hf_ptq.py @@ -328,6 +328,9 @@ def main(args): model = model.language_model model_type = get_model_type(model) + if model_type == "phi4mm": + warnings.warn("Please set the default input_mode to InputMode.LANGUAGE before quantizing.") + if args.sparsity_fmt != "dense": if args.batch_size == 0: # Sparse algorithm takes more GPU memory so we reduce the batch_size by 4. @@ -478,9 +481,6 @@ def main(args): quant_cfg["quant_cfg"]["*audio*"] = {"enable": False} quant_cfg["quant_cfg"]["*image*"] = {"enable": False} quant_cfg["quant_cfg"]["*vision*"] = {"enable": False} - warnings.warn( - "Please set the default input_mode to InputMode.LANGUAGE before quantizing." - ) if not model_is_already_quantized or calibration_only: # Only run single sample for preview