2/5 support kimi 2.5 full + lora: VL-aware quantization/conversion tools (#1220)

Co-authored-by: JiLi <22428217+GeLee-Q@users.noreply.github.com>
This commit is contained in:
Nan Jiang
2026-05-28 14:46:27 -07:00
committed by GitHub
co-authored by JiLi
parent 5af8043da7
commit 210b839b1d
4 changed files with 16 additions and 6 deletions
+2
View File
@@ -137,6 +137,8 @@ def process_file(input_path, output_path, filename, strategy, block_size, result
and "lm_head" not in key
and "eh_proj" not in key
and "weights_proj" not in key
and "vision_tower" not in key
and "mm_projector" not in key
):
qw, s = quant_fp8(weights[key], strategy, block_size)
q_weights[key] = qw
+2
View File
@@ -79,6 +79,8 @@ def main():
"re:.*shared_experts.*",
"re:.*mlp\\.(gate|up|gate_up|down)_proj.*",
"re:.*mlp\\.gate\\.*",
"re:vision_tower.*",
"re:mm_projector.*",
]
recipe = GPTQModifier(
+2
View File
@@ -295,6 +295,8 @@ def parse_args():
"re:.*shared_experts.*",
"re:.*mlp\\.(gate|up|gate_up|down)_proj.*",
"re:.*mlp\\.gate\\.*",
"re:vision_tower.*",
"re:mm_projector.*",
],
help="Ignore Rules",
)
@@ -1,7 +1,7 @@
"""
Usage:
------
python convert_k2_thinking_int4_to_bf16.py [-h] --model-dir MODEL_DIR [--output-dir OUTPUT_DIR]
python convert_kimi_int4_to_bf16.py [-h] --model-dir MODEL_DIR [--output-dir OUTPUT_DIR]
[--files FILE [FILE ...]] [--config-path CONFIG_PATH]
[--overwrite]
options:
@@ -20,7 +20,8 @@ options:
Example:
--------
python convert_k2_thinking_int4_to_bf16.py --model-dir /Kimi-K2-Thinking --output-dir /Kimi-K2-Thinking-bf16
python convert_kimi_int4_to_bf16.py --model-dir /Kimi-K2-Thinking --output-dir /Kimi-K2-Thinking-bf16
python convert_kimi_int4_to_bf16.py --model-dir /Kimi-K2.5 --output-dir /Kimi-K2.5-bf16
"""
import argparse
@@ -40,10 +41,11 @@ def _load_config(model_dir: str, config_path: str | None) -> tuple[int, int, int
cfg_path = config_path or os.path.join(model_dir, "config.json")
with open(cfg_path) as f:
cfg = json.load(f)
hidden_size = int(cfg.get("hidden_size"))
inter_size = int(cfg.get("moe_intermediate_size"))
loader = cfg.get("text_config", cfg)
hidden_size = int(loader.get("hidden_size"))
inter_size = int(loader.get("moe_intermediate_size"))
group_size = int(
cfg.get("quantization_config", {})
loader.get("quantization_config", {})
.get("config_groups", {})
.get("group_0", {})
.get("weights", {})
@@ -216,9 +218,11 @@ def main():
for fname in os.listdir(model_dir):
src_path = os.path.join(model_dir, fname)
dst_path = os.path.join(output_dir, fname)
extensions_to_copy = (".json", ".py", ".jinja", ".model")
if fname == "model.safetensors.index.json":
continue
if fname.endswith(".json") or fname.endswith(".py") or fname.startswith("tokenizer"):
if fname.startswith("tokenizer") or any(fname.endswith(ext) for ext in extensions_to_copy):
shutil.copy2(src_path, dst_path)
# Generate new index