mirror of
https://github.com/radixark/miles.git
synced 2026-10-02 07:14:53 +08:00
2/5 support kimi 2.5 full + lora: VL-aware quantization/conversion tools (#1220)
Co-authored-by: JiLi <22428217+GeLee-Q@users.noreply.github.com>
This commit is contained in:
@@ -137,6 +137,8 @@ def process_file(input_path, output_path, filename, strategy, block_size, result
|
||||
and "lm_head" not in key
|
||||
and "eh_proj" not in key
|
||||
and "weights_proj" not in key
|
||||
and "vision_tower" not in key
|
||||
and "mm_projector" not in key
|
||||
):
|
||||
qw, s = quant_fp8(weights[key], strategy, block_size)
|
||||
q_weights[key] = qw
|
||||
|
||||
@@ -79,6 +79,8 @@ def main():
|
||||
"re:.*shared_experts.*",
|
||||
"re:.*mlp\\.(gate|up|gate_up|down)_proj.*",
|
||||
"re:.*mlp\\.gate\\.*",
|
||||
"re:vision_tower.*",
|
||||
"re:mm_projector.*",
|
||||
]
|
||||
|
||||
recipe = GPTQModifier(
|
||||
|
||||
@@ -295,6 +295,8 @@ def parse_args():
|
||||
"re:.*shared_experts.*",
|
||||
"re:.*mlp\\.(gate|up|gate_up|down)_proj.*",
|
||||
"re:.*mlp\\.gate\\.*",
|
||||
"re:vision_tower.*",
|
||||
"re:mm_projector.*",
|
||||
],
|
||||
help="Ignore Rules",
|
||||
)
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""
|
||||
Usage:
|
||||
------
|
||||
python convert_k2_thinking_int4_to_bf16.py [-h] --model-dir MODEL_DIR [--output-dir OUTPUT_DIR]
|
||||
python convert_kimi_int4_to_bf16.py [-h] --model-dir MODEL_DIR [--output-dir OUTPUT_DIR]
|
||||
[--files FILE [FILE ...]] [--config-path CONFIG_PATH]
|
||||
[--overwrite]
|
||||
options:
|
||||
@@ -20,7 +20,8 @@ options:
|
||||
|
||||
Example:
|
||||
--------
|
||||
python convert_k2_thinking_int4_to_bf16.py --model-dir /Kimi-K2-Thinking --output-dir /Kimi-K2-Thinking-bf16
|
||||
python convert_kimi_int4_to_bf16.py --model-dir /Kimi-K2-Thinking --output-dir /Kimi-K2-Thinking-bf16
|
||||
python convert_kimi_int4_to_bf16.py --model-dir /Kimi-K2.5 --output-dir /Kimi-K2.5-bf16
|
||||
"""
|
||||
|
||||
import argparse
|
||||
@@ -40,10 +41,11 @@ def _load_config(model_dir: str, config_path: str | None) -> tuple[int, int, int
|
||||
cfg_path = config_path or os.path.join(model_dir, "config.json")
|
||||
with open(cfg_path) as f:
|
||||
cfg = json.load(f)
|
||||
hidden_size = int(cfg.get("hidden_size"))
|
||||
inter_size = int(cfg.get("moe_intermediate_size"))
|
||||
loader = cfg.get("text_config", cfg)
|
||||
hidden_size = int(loader.get("hidden_size"))
|
||||
inter_size = int(loader.get("moe_intermediate_size"))
|
||||
group_size = int(
|
||||
cfg.get("quantization_config", {})
|
||||
loader.get("quantization_config", {})
|
||||
.get("config_groups", {})
|
||||
.get("group_0", {})
|
||||
.get("weights", {})
|
||||
@@ -216,9 +218,11 @@ def main():
|
||||
for fname in os.listdir(model_dir):
|
||||
src_path = os.path.join(model_dir, fname)
|
||||
dst_path = os.path.join(output_dir, fname)
|
||||
extensions_to_copy = (".json", ".py", ".jinja", ".model")
|
||||
|
||||
if fname == "model.safetensors.index.json":
|
||||
continue
|
||||
if fname.endswith(".json") or fname.endswith(".py") or fname.startswith("tokenizer"):
|
||||
if fname.startswith("tokenizer") or any(fname.endswith(ext) for ext in extensions_to_copy):
|
||||
shutil.copy2(src_path, dst_path)
|
||||
|
||||
# Generate new index
|
||||
Reference in New Issue
Block a user