mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
### What does this PR do? Type of change: new feature Packages the existing ModelOpt agent skills as installable Codex and Claude plugins: - Adds a repo-scoped Codex marketplace and Claude-compatible marketplace. - Adds the canonical `plugins/modelopt/` plugin tree and manifests. - Moves the skill tree into the plugin and keeps `.agents/skills` as a compatibility symlink. - Adds a minimal `common` placeholder skill required by Codex validation. - Documents installation from this repository. ### Usage ```bash codex plugin marketplace add NVIDIA/Model-Optimizer ``` Then open `/plugins`, select the `modelopt` marketplace, and install `modelopt`. For Claude Code: ```bash claude plugin marketplace add https://github.com/NVIDIA/Model-Optimizer.git claude plugin install modelopt@modelopt ``` ### Testing - Codex plugin validator - `claude plugin validate . --strict` - `claude plugin validate plugins/modelopt --strict` 1. Install the marketplace plugin with Codex and Claude from an unrelated temporary workspace. 2. Exercise packaged evaluation helpers, a day-0 gate, and the shared remote helper from that workspace. 3. Run `uv run --frozen --extra dev python -m pytest -q plugins/modelopt/skills/day0-release/tests/test_gates.py plugins/modelopt/skills/benchmark-model-kernels/tests`. 4. Run pre-commit hooks for all changed files. ### Before your PR is "*Ready for review*" - Is this change backward compatible?: ✅ - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: N/A - Did you write any new necessary tests?: ✅ — added a plugin-path validator; existing focused skill tests and installed-plugin smoke tests pass. - Did you update Changelog?: N/A — agent tooling and distribution only. - Did you get Claude approval on this PR?: N/A ### Additional Information Skills remain available through `.agents/skills`; bundled helpers are packaged under the plugin and resolved from `$SKILL_DIR` so installed workflows do not depend on the current workspace. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added installable ModelOpt plugins for Claude Code and Codex. * Added skills for PTQ, deployment, evaluation, monitoring, debugging, benchmarking, MLflow access, EAGLE3 workflows, and release management. * Added deployment helpers, evaluation recipes, checkpoint validation, and release-gating tools. * **Documentation** * Expanded setup, credential, SLURM, benchmarking, deployment, evaluation, troubleshooting, and workspace guidance. * Added installation instructions and updated agent-skill discovery guidance. * **Maintenance** * Updated skill references and compatibility links for reliable use across supported plugin environments. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Chad Voegele <cvoegele@nvidia.com>
114 lines
7.3 KiB
YAML
114 lines
7.3 KiB
YAML
# nel-next (nemo-evaluator 0.4.x) agentic AA eval template — NOT the 0.2.6 schema.
|
|
# Read references/nel-next.md + recipes/tasks/aa_next/<benchmark>.md first.
|
|
# Benchmark shown = Terminal-Bench 2.1; swap the `benchmarks:` block per the recipe.
|
|
#
|
|
# Run via the isolated nel-next venv:
|
|
# "$SKILL_DIR/scripts/nel-next.sh" --setup-only
|
|
# set -a && source .env && set +a # HF_TOKEN, AWS_*, NEL_NEXT_EVAL_IMAGE, HARBOR_*_ECR_REPOSITORY (from modelopttools:eval-config)
|
|
# "$SKILL_DIR/scripts/nel-next.sh" eval run "$SKILL_DIR/recipes/examples/example_eval_next.yaml" --dry-run
|
|
# ... --submit -O benchmarks.0.max_problems=2 -O benchmarks.0.repeats=1 -O benchmarks.0.max_concurrent=2 # canary
|
|
# ... --submit # full
|
|
# Internal harbor infra (eval_image + ECR) comes from .env via ${VAR}; all blocks
|
|
# validate against the pinned nel-next build (extra="forbid"). `???` = fill.
|
|
|
|
services:
|
|
model: # any name; referenced by benchmarks[].solver.service
|
|
type: vllm # NEL deploys + serves (auto-wrapped as an api service for the agent)
|
|
model: ??? # checkpoint path (bind-mounted to /model:ro) OR HF handle
|
|
served_model_name: ???
|
|
protocol: chat_completions
|
|
port: 5000
|
|
tensor_parallel_size: 8 # STRUCTURED fields — don't repeat parallelism in extra_args
|
|
data_parallel_size: 1
|
|
num_nodes: 1
|
|
image: vllm/vllm-openai:v0.26.0 # vLLM SERVING image (≠ eval_image); bump to the model's recipes.vllm.ai min
|
|
startup_timeout: 3600.0
|
|
extra_args: # raw vllm flags (cross-check model card + recipes.vllm.ai)
|
|
- "--trust-remote-code"
|
|
- "--enable-auto-tool-choice" # agentic = tool-calling
|
|
- "--tool-call-parser=???" # REQUIRED — model-specific (e.g. minimax_m2, glm47); pick per the chat template
|
|
# - "--reasoning-parser=???" # reasoning models only (e.g. minimax_m2_append_think, glm45)
|
|
- "--enable-expert-parallel" # MoE only
|
|
- "--enable-prefix-caching"
|
|
- "--gpu-memory-utilization=0.9"
|
|
- "--max-num-seqs=64"
|
|
- '--model-loader-extra-config={"enable_multithread_load":true,"num_threads":48}'
|
|
extra_env: # carry the model's normal backend env (e.g. NVFP4 MoE: VLLM_USE_FLASHINFER_MOE_FP4=1, VLLM_FLASHINFER_MOE_BACKEND=throughput)
|
|
VLLM_CACHE_ROOT: /cache/vllm
|
|
HF_HOME: /cache/huggingface
|
|
HF_TOKEN: ${HF_TOKEN}
|
|
SAFETENSORS_FAST_GPU: "1"
|
|
container_mounts: # source dirs MUST pre-exist (pyxis won't create them): ssh <login> 'mkdir -p <lustre>/<user>/.cache/{vllm,huggingface}'
|
|
- ???:/cache/vllm
|
|
- ???:/cache/huggingface
|
|
generation: {temperature: 1.0, top_p: 0.95} # from model card (reasoning mode); adjust per card — mandatory lookup (references/model-card-research.md), same as 0.2.6
|
|
proxy:
|
|
request_timeout: 3600 # canonical; MUST be >= benchmarks[].solver.agent_kwargs.llm_kwargs.timeout
|
|
extra_body: {skip_special_tokens: false} # add model-card sampling extras here if the card specifies them; mirror them in the export tags below
|
|
interceptors:
|
|
- name: drop_params # agents send max_tokens; many servers reject it
|
|
# last two are sent by the 0.5.x harbor eval image; vLLM 400s on them unless stripped
|
|
config: {params: [max_tokens, max_completion_tokens, max_input_tokens_per_task, no_rebuild]}
|
|
# SWE-bench (OpenHands, multi-turn) adds turn_counter + system_message AND USES A DIFFERENT
|
|
# ORDER (drop_params before consolidate_system) — don't lift this chain — see swebench_verified.md
|
|
# FEP-1104/1120 diagnostics — uncomment for a CANARY/debug run, drop it for the scored run:
|
|
# first_n caps only 200s, so every error pair (full req+res bodies) is retained in memory for
|
|
# the whole run and re-serialized on each write — unbounded growth exactly when the server errors.
|
|
# - name: http_pairs_dump # canonical LAST in the chain (SWE-bench: first)
|
|
# config: {dump_path: "$${NEL_OUTPUT_DIR}/http_pairs_metrics.json", first_n: 50} # $$ defers expansion to run time
|
|
node_pool: gpu
|
|
|
|
benchmarks:
|
|
- playbook: terminal_bench_2_1 # SWE-bench: swebench_verified (see recipe)
|
|
repeats: 8 # AA count — keep for a scored run (canary: -O benchmarks.0.repeats=1)
|
|
max_concurrent: 50 # canonical TB2.1 bench.yaml; keep == sandbox.concurrency
|
|
solver:
|
|
service: model
|
|
timeout_strategy: max # canonical TB2.1; "task" = leaderboard-comparable
|
|
agent_kwargs: {llm_kwargs: {timeout: 3600}}
|
|
sandbox:
|
|
region: us-east-1 # MUST match the region in ${HARBOR_ECR_REPOSITORY} (SWE-bench: us-east-2 + ${HARBOR_SWEBENCH_ECR_REPOSITORY})
|
|
ecr_repository: ${HARBOR_ECR_REPOSITORY} # from modelopttools:eval-config
|
|
concurrency: 50
|
|
log_stream_prefix: terminalbench21-???
|
|
|
|
cluster:
|
|
type: slurm
|
|
hostname: ??? # login FQDN
|
|
username: ${oc.env:USER}
|
|
account: ???
|
|
walltime: "04:00:00" # auto_resume chains across windows
|
|
shards: 1 # N = N nodes (each redeploys vLLM)
|
|
eval_image: ${NEL_NEXT_EVAL_IMAGE} # from eval-config (0.5.0.1-harbor, multi-arch; enroot creds per SKILL Step 7.5)
|
|
sbatch_comment: '{"OccupiedIdleGPUsJobReaper":{"exemptIdleTimeMins":"480","reason":"benchmarking","description":"nel-next agentic eval"}}'
|
|
sbatch_extra_flags: {switches: 1, exclusive: true}
|
|
container_env: # AWS creds reach the eval container ONLY via here
|
|
HF_TOKEN: ${HF_TOKEN}
|
|
HF_HOME: /cache/huggingface
|
|
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
|
|
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
|
|
AWS_DEFAULT_REGION: us-east-1 # match sandbox.region / the region in ${HARBOR_ECR_REPOSITORY}
|
|
LLM_API_KEY: "no-key-needed"
|
|
mount_home: false
|
|
auto_resume: true
|
|
max_retries: 3
|
|
node_pools:
|
|
gpu: {partition: "???", nodes: 1, ntasks_per_node: 1, gpus_per_node: 8} # match TP*DP
|
|
|
|
output:
|
|
dir: ??? # <lustre>/eval-output/<model>-<benchmark>
|
|
# MLflow export config (experiment/tags/description). SLURM does NOT auto-export —
|
|
# push after the run finishes: `nel-next.sh mlflow-push -r <run_id> -c <this>.yaml`
|
|
# (reads this block, stages the bundle off the cluster, exports with traces off).
|
|
export: [mlflow]
|
|
export_config:
|
|
mlflow:
|
|
tracking_uri: ${MLFLOW_TRACKING_URI} # from modelopttools:eval-config (canonical mlflow.frontier-evals host; NOT the -nemo-evaluator alias)
|
|
experiment_name: ??? # <user>/<model>-<benchmark> (hardcode; ${USER}=root in-container)
|
|
log_config_params: true
|
|
copy_logs: true
|
|
exclude_patterns: ["shard*", "model_traffic.jsonl"] # captured request bodies (FEA-224) stay in the run dir
|
|
description: ??? # '<model> | T=1.0 top_p=0.95 | <benchmark> (timeout_strategy=…) | r8'
|
|
# model/checkpoint_path/benchmark drive dashboard attribution (engine logs only a generic metric key); temperature/top_p mirror generation above.
|
|
tags: {framework: vllm, model: "???", checkpoint_path: "???", benchmark: "???", temperature: '1.0', top_p: '0.95'}
|