From 210b839b1d3d670362b77d1237d6a2e07a6c9abf Mon Sep 17 00:00:00 2001 From: Nan Jiang <59716405+nanjiangwill@users.noreply.github.com> Date: Thu, 28 May 2026 17:46:27 -0400 Subject: [PATCH] 2/5 support kimi 2.5 full + lora: VL-aware quantization/conversion tools (#1220) Co-authored-by: JiLi <22428217+GeLee-Q@users.noreply.github.com> --- tools/convert_hf_to_fp8.py | 2 ++ tools/convert_hf_to_int4.py | 2 ++ tools/convert_hf_to_int4_direct.py | 2 ++ ...4_to_bf16.py => convert_kimi_int4_to_bf16.py} | 16 ++++++++++------ 4 files changed, 16 insertions(+), 6 deletions(-) rename tools/{convert_k2_thinking_int4_to_bf16.py => convert_kimi_int4_to_bf16.py} (92%) diff --git a/tools/convert_hf_to_fp8.py b/tools/convert_hf_to_fp8.py index 7754e7dea8..4043025e54 100644 --- a/tools/convert_hf_to_fp8.py +++ b/tools/convert_hf_to_fp8.py @@ -137,6 +137,8 @@ def process_file(input_path, output_path, filename, strategy, block_size, result and "lm_head" not in key and "eh_proj" not in key and "weights_proj" not in key + and "vision_tower" not in key + and "mm_projector" not in key ): qw, s = quant_fp8(weights[key], strategy, block_size) q_weights[key] = qw diff --git a/tools/convert_hf_to_int4.py b/tools/convert_hf_to_int4.py index c5620b071f..0a8d14f649 100644 --- a/tools/convert_hf_to_int4.py +++ b/tools/convert_hf_to_int4.py @@ -79,6 +79,8 @@ def main(): "re:.*shared_experts.*", "re:.*mlp\\.(gate|up|gate_up|down)_proj.*", "re:.*mlp\\.gate\\.*", + "re:vision_tower.*", + "re:mm_projector.*", ] recipe = GPTQModifier( diff --git a/tools/convert_hf_to_int4_direct.py b/tools/convert_hf_to_int4_direct.py index e741b802d3..2fa28e1aab 100644 --- a/tools/convert_hf_to_int4_direct.py +++ b/tools/convert_hf_to_int4_direct.py @@ -295,6 +295,8 @@ def parse_args(): "re:.*shared_experts.*", "re:.*mlp\\.(gate|up|gate_up|down)_proj.*", "re:.*mlp\\.gate\\.*", + "re:vision_tower.*", + "re:mm_projector.*", ], help="Ignore Rules", ) diff --git a/tools/convert_k2_thinking_int4_to_bf16.py b/tools/convert_kimi_int4_to_bf16.py similarity index 92% rename from tools/convert_k2_thinking_int4_to_bf16.py rename to tools/convert_kimi_int4_to_bf16.py index 78c0effc04..3614035432 100644 --- a/tools/convert_k2_thinking_int4_to_bf16.py +++ b/tools/convert_kimi_int4_to_bf16.py @@ -1,7 +1,7 @@ """ Usage: ------ -python convert_k2_thinking_int4_to_bf16.py [-h] --model-dir MODEL_DIR [--output-dir OUTPUT_DIR] +python convert_kimi_int4_to_bf16.py [-h] --model-dir MODEL_DIR [--output-dir OUTPUT_DIR] [--files FILE [FILE ...]] [--config-path CONFIG_PATH] [--overwrite] options: @@ -20,7 +20,8 @@ options: Example: -------- -python convert_k2_thinking_int4_to_bf16.py --model-dir /Kimi-K2-Thinking --output-dir /Kimi-K2-Thinking-bf16 +python convert_kimi_int4_to_bf16.py --model-dir /Kimi-K2-Thinking --output-dir /Kimi-K2-Thinking-bf16 +python convert_kimi_int4_to_bf16.py --model-dir /Kimi-K2.5 --output-dir /Kimi-K2.5-bf16 """ import argparse @@ -40,10 +41,11 @@ def _load_config(model_dir: str, config_path: str | None) -> tuple[int, int, int cfg_path = config_path or os.path.join(model_dir, "config.json") with open(cfg_path) as f: cfg = json.load(f) - hidden_size = int(cfg.get("hidden_size")) - inter_size = int(cfg.get("moe_intermediate_size")) + loader = cfg.get("text_config", cfg) + hidden_size = int(loader.get("hidden_size")) + inter_size = int(loader.get("moe_intermediate_size")) group_size = int( - cfg.get("quantization_config", {}) + loader.get("quantization_config", {}) .get("config_groups", {}) .get("group_0", {}) .get("weights", {}) @@ -216,9 +218,11 @@ def main(): for fname in os.listdir(model_dir): src_path = os.path.join(model_dir, fname) dst_path = os.path.join(output_dir, fname) + extensions_to_copy = (".json", ".py", ".jinja", ".model") + if fname == "model.safetensors.index.json": continue - if fname.endswith(".json") or fname.endswith(".py") or fname.startswith("tokenizer"): + if fname.startswith("tokenizer") or any(fname.endswith(ext) for ext in extensions_to_copy): shutil.copy2(src_path, dst_path) # Generate new index