feat(pricing): real per-model cost catalog; fix non-Claude mispricing

Replaces the hardcoded 5-model Claude price table (extract.ts) with a curated, websearch-sourced catalog of 63 models across Anthropic/OpenAI/Google/Chinese/other vendors, behind a small lookup/compute module (src/pricing/catalog.ts). Non-Claude models now price off their own rows instead of defaulting to Claude Sonnet. Adds provider/ prefix normalization, native-cost passthrough, and null+warn on unknown models. LiteLLM ~2900-model catalog kept as a dev-only refresh base (tools/pricing/, gitignored), not bundled.
This commit is contained in:
Mert Koseoglu
2026-06-26 00:35:26 +03:00
parent 4112e8485b
commit 82a8d0cfac
11 changed files with 1116 additions and 49 deletions
+4
View File
@@ -39,5 +39,9 @@ context-mode-guidance-*/
.claude-worktrees/
.vibetree/
.cw/
# Dev-only pricing source: the full litellm catalog (~1.5MB) is the manual
# refresh base for src/pricing/sources/*.json — never bundled into the plugin.
# Refresh: see tools/pricing/litellm-NOTES.md. Keep NOTES tracked, ignore the blob.
tools/pricing/litellm-catalog.json
/.cocoindex_code/
/.kilo/
+203
View File
@@ -0,0 +1,203 @@
/**
* Pricing catalog — single source of truth for per-model USD cost.
*
* Deep module, tiny interface. At load it merges the 5 curated vendor JSONs
* (src/pricing/sources/{anthropic,openai,google,chinese,others}.json) into one
* Map<modelId, Price> in per-Mtok units, then exposes three pure-ish functions:
*
* lookupPrice(modelId) → Price | null
* computeCostUsd(modelId, tokens) → number | null
* nativeOrComputed(id, t, native) → number | null
*
* WHY THIS EXISTS — the bug it kills:
* The old table in src/session/extract.ts hardcoded ~5 Claude rows plus a
* `default` row, and any unmatched id (every OpenAI / Gemini / Qwen / DeepSeek
* / Grok model) silently inherited Claude-Sonnet pricing. Non-Claude turns were
* therefore mispriced. Here each model is priced from ITS OWN curated row, and
* an unknown id resolves to `null` (no price) instead of a wrong Claude rate.
*
* The large litellm catalog (~1.5MB, ~2900 models) is NOT bundled — it lives at
* tools/pricing/litellm-catalog.json as the dev-only refresh base for these
* curated JSONs. See tools/pricing/litellm-NOTES.md for the refresh recipe.
*
* The 5 curated JSONs are small (~25KB total) and esbuild inlines them into the
* hook/server bundles at build time (no runtime fs read, no external file).
*/
import anthropic from "./sources/anthropic.json" with { type: "json" };
import openai from "./sources/openai.json" with { type: "json" };
import google from "./sources/google.json" with { type: "json" };
import chinese from "./sources/chinese.json" with { type: "json" };
import others from "./sources/others.json" with { type: "json" };
/** Per-Mtok price for one model. Any of the four rates may be null ("unknown"). */
export interface Price {
input_per_mtok: number | null;
output_per_mtok: number | null;
cache_read_per_mtok: number | null;
cache_write_per_mtok: number | null;
}
/** Token counts for one turn. All fields optional; absent ⇒ treated as 0. */
export interface TokenCounts {
input_tokens?: number;
output_tokens?: number;
cache_read_tokens?: number;
cache_creation_tokens?: number;
}
/** Raw shape of a curated source row (carries provenance fields we drop). */
interface RawRow {
input_per_mtok: number | null;
output_per_mtok: number | null;
cache_read_per_mtok: number | null;
cache_write_per_mtok: number | null;
[extra: string]: unknown;
}
/**
* Merge the 5 vendor JSONs into one Map. A row with a null *input* price is
* unusable for cost (the primary bucket has no rate) and is dropped at load so
* lookupPrice returns null for it — matching "null-priced entries → no price".
* Curated ids are globally unique across the five files (verified), so merge
* order is irrelevant; later files would otherwise win on collision.
*/
function buildCatalog(): Map<string, Price> {
const map = new Map<string, Price>();
const sources: Record<string, RawRow>[] = [
anthropic as Record<string, RawRow>,
openai as Record<string, RawRow>,
google as Record<string, RawRow>,
chinese as Record<string, RawRow>,
others as Record<string, RawRow>,
];
for (const src of sources) {
for (const id of Object.keys(src)) {
const row = src[id];
if (row == null || typeof row !== "object") continue;
// No input rate ⇒ no usable price for this model.
if (typeof row.input_per_mtok !== "number") continue;
map.set(id, {
input_per_mtok: row.input_per_mtok,
output_per_mtok: typeof row.output_per_mtok === "number" ? row.output_per_mtok : null,
cache_read_per_mtok:
typeof row.cache_read_per_mtok === "number" ? row.cache_read_per_mtok : null,
cache_write_per_mtok:
typeof row.cache_write_per_mtok === "number" ? row.cache_write_per_mtok : null,
});
}
}
return map;
}
const CATALOG: Map<string, Price> = buildCatalog();
/**
* Strip a single leading `provider/` segment, char-algorithmically (NO regex).
* Walks to the first '/'; everything after it is the bare model id. Only the
* FIRST segment is stripped — `openai/gpt-5` → `gpt-5`, `a/b/c` → `b/c` — so a
* model id that legitimately contains a slash keeps its remaining segments.
* Returns null when there is no '/' (caller already tried the raw form).
*/
function stripProviderPrefix(id: string): string | null {
for (let i = 0; i < id.length; i++) {
if (id.charCodeAt(i) === 47 /* '/' */) {
// Guard against a leading or trailing slash producing an empty segment.
if (i === 0 || i === id.length - 1) return null;
return id.slice(i + 1);
}
}
return null;
}
/** trim + lowercase, char-safe (String.prototype.trim/toLowerCase, no regex). */
function normalize(id: string): string {
return id.trim().toLowerCase();
}
/**
* Resolve a model id to its curated Price, or null on miss.
* Strategy: exact, then normalized (trim+lowercase), then provider-stripped
* normalized. Misses return null so the caller can decide (warn / fall back).
*/
export function lookupPrice(modelId: string): Price | null {
if (typeof modelId !== "string" || modelId.length === 0) return null;
// 1. Exact — fastest path, covers ids already in canonical form.
const exact = CATALOG.get(modelId);
if (exact) return exact;
// 2. Normalized — trimmed + lowercased.
const norm = normalize(modelId);
const byNorm = CATALOG.get(norm);
if (byNorm) return byNorm;
// 3. Provider-stripped (pi / openclaw / opencode report `provider/model`).
const bare = stripProviderPrefix(norm);
if (bare) {
const byBare = CATALOG.get(bare);
if (byBare) return byBare;
}
return null;
}
/** Price one token bucket. A null bucket rate falls back to the input rate. */
function bucketCost(tokens: number, rate: number | null, inputRate: number): number {
if (tokens <= 0) return 0;
const effective = typeof rate === "number" ? rate : inputRate;
return tokens * effective;
}
/**
* Σ tokens × per-Mtok price / 1e6 over the four buckets. Returns null when no
* price is found OR every token count is zero/absent (so dashboards never show
* a misleading "$0.00 for nothing" row). On a price miss, warns exactly once
* with the unmatched id so the curated catalog can be extended.
*
* Bucket → price mapping:
* input_tokens → input_per_mtok
* output_tokens → output_per_mtok (null ⇒ input rate)
* cache_read_tokens → cache_read_per_mtok (null ⇒ input rate)
* cache_creation_tokens → cache_write_per_mtok (null ⇒ input rate)
*/
export function computeCostUsd(modelId: string, t: TokenCounts): number | null {
const input = typeof t.input_tokens === "number" ? t.input_tokens : 0;
const output = typeof t.output_tokens === "number" ? t.output_tokens : 0;
const cacheRead = typeof t.cache_read_tokens === "number" ? t.cache_read_tokens : 0;
const cacheCreate =
typeof t.cache_creation_tokens === "number" ? t.cache_creation_tokens : 0;
// All buckets empty ⇒ nothing to price, regardless of model.
if (input <= 0 && output <= 0 && cacheRead <= 0 && cacheCreate <= 0) return null;
const price = lookupPrice(modelId);
if (!price || typeof price.input_per_mtok !== "number") {
// Unknown model — emit one line so the id can be added to the catalog.
console.warn(`[pricing] no curated price for model id: ${modelId}`);
return null;
}
const inputRate = price.input_per_mtok;
const microDollars =
bucketCost(input, inputRate, inputRate) +
bucketCost(output, price.output_per_mtok, inputRate) +
bucketCost(cacheRead, price.cache_read_per_mtok, inputRate) +
bucketCost(cacheCreate, price.cache_write_per_mtok, inputRate);
return microDollars / 1_000_000;
}
/**
* Prefer a provider-supplied native cost when present, else compute from the
* catalog. A native cost of exactly 0 is a real value (free tier) and passes
* through — only null/undefined defers to computeCostUsd.
*/
export function nativeOrComputed(
modelId: string,
t: TokenCounts,
nativeCostUsd?: number | null,
): number | null {
if (typeof nativeCostUsd === "number") return nativeCostUsd;
return computeCostUsd(modelId, t);
}
+82
View File
@@ -0,0 +1,82 @@
{
"claude-opus-4-8": {
"input_per_mtok": 5.00,
"output_per_mtok": 25.00,
"cache_read_per_mtok": 0.50,
"cache_write_per_mtok": 6.25,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-opus-4-7": {
"input_per_mtok": 5.00,
"output_per_mtok": 25.00,
"cache_read_per_mtok": 0.50,
"cache_write_per_mtok": 6.25,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-opus-4-6": {
"input_per_mtok": 5.00,
"output_per_mtok": 25.00,
"cache_read_per_mtok": 0.50,
"cache_write_per_mtok": 6.25,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-opus-4-5": {
"input_per_mtok": 5.00,
"output_per_mtok": 25.00,
"cache_read_per_mtok": 0.50,
"cache_write_per_mtok": 6.25,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-sonnet-4-6": {
"input_per_mtok": 3.00,
"output_per_mtok": 15.00,
"cache_read_per_mtok": 0.30,
"cache_write_per_mtok": 3.75,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-sonnet-4-5": {
"input_per_mtok": 3.00,
"output_per_mtok": 15.00,
"cache_read_per_mtok": 0.30,
"cache_write_per_mtok": 3.75,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-haiku-4-5": {
"input_per_mtok": 1.00,
"output_per_mtok": 5.00,
"cache_read_per_mtok": 0.10,
"cache_write_per_mtok": 1.25,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-3-7-sonnet": {
"input_per_mtok": 3.00,
"output_per_mtok": 15.00,
"cache_read_per_mtok": 0.30,
"cache_write_per_mtok": 3.75,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"claude-3-5-haiku": {
"input_per_mtok": 0.80,
"output_per_mtok": 4.00,
"cache_read_per_mtok": 0.08,
"cache_write_per_mtok": 1.00,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
},
"claude-fable-5": {
"input_per_mtok": 10.00,
"output_per_mtok": 50.00,
"cache_read_per_mtok": 1.00,
"cache_write_per_mtok": 12.50,
"source": "https://platform.claude.com/docs/en/about-claude/pricing",
"as_of": "2026-06"
}
}
+155
View File
@@ -0,0 +1,155 @@
{
"qwen3-coder": {
"input_per_mtok": 1.0,
"output_per_mtok": 5.0,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"as_of": "2026-06",
"_note": "Billing id maps to qwen3-coder-plus. Tiered by input length; headline is the 0-32K tier. Higher tiers: 32K-128K in 1.8/out 9, 128K-256K in 3/out 15, 256K-1M in 6/out 60. Cross-checked vs LiteLLM tiered_pricing."
},
"qwen-max": {
"input_per_mtok": 1.6,
"output_per_mtok": 6.4,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"as_of": "2026-06",
"_note": "Alibaba DashScope international (USD-denominated). Cross-checked vs LiteLLM dashscope/qwen-max."
},
"qwen-plus": {
"input_per_mtok": 0.4,
"output_per_mtok": 1.2,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"as_of": "2026-06",
"_note": "Alibaba DashScope international (USD). Cross-checked vs LiteLLM dashscope/qwen-plus."
},
"qwen-turbo": {
"input_per_mtok": 0.05,
"output_per_mtok": 0.2,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"as_of": "2026-06",
"_note": "Alibaba DashScope international (USD). Cross-checked vs LiteLLM dashscope/qwen-turbo."
},
"qwen3-max": {
"input_per_mtok": 1.2,
"output_per_mtok": 6.0,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"as_of": "2026-06",
"_note": "Tiered by input length; headline is the 0-32K tier. Higher tiers: 32K-128K in 2.4/out 12, 128K-252K in 3/out 15. Cross-checked vs LiteLLM dashscope/qwen3-max tiered_pricing."
},
"kimi-k2": {
"input_per_mtok": 0.6,
"output_per_mtok": 2.5,
"cache_read_per_mtok": 0.15,
"cache_write_per_mtok": null,
"source": "https://platform.moonshot.ai/docs/pricing/chat",
"as_of": "2026-06",
"_note": "Moonshot international platform (USD). Resolves to kimi-k2-0905-preview. Cache-hit input 0.15. Cross-checked vs LiteLLM moonshot/kimi-k2-0905-preview."
},
"kimi-k2-turbo": {
"input_per_mtok": 1.15,
"output_per_mtok": 8.0,
"cache_read_per_mtok": 0.15,
"cache_write_per_mtok": null,
"source": "https://platform.moonshot.ai/docs/pricing/chat",
"as_of": "2026-06",
"_note": "Moonshot international (USD). Resolves to kimi-k2-turbo-preview. Cache-hit input 0.15. Cross-checked vs LiteLLM moonshot/kimi-k2-turbo-preview."
},
"moonshot-v1-8k": {
"input_per_mtok": 0.2,
"output_per_mtok": 2.0,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://platform.moonshot.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Moonshot international (USD). Cross-checked vs LiteLLM moonshot/moonshot-v1-8k."
},
"moonshot-v1-32k": {
"input_per_mtok": 1.0,
"output_per_mtok": 3.0,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://platform.moonshot.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Moonshot international (USD). Cross-checked vs LiteLLM moonshot/moonshot-v1-32k."
},
"moonshot-v1-128k": {
"input_per_mtok": 2.0,
"output_per_mtok": 5.0,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://platform.moonshot.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Moonshot international (USD). Cross-checked vs LiteLLM moonshot/moonshot-v1-128k."
},
"deepseek-v3": {
"input_per_mtok": 0.27,
"output_per_mtok": 1.1,
"cache_read_per_mtok": 0.07,
"cache_write_per_mtok": 0.0,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"as_of": "2026-06",
"_note": "Legacy DeepSeek-V3 era pricing. No longer shown on the official current pricing page (which now lists deepseek-v4-flash/pro). Value from LiteLLM deepseek/deepseek-v3, last sourced from the official page. cache-hit input 0.07."
},
"deepseek-r1": {
"input_per_mtok": 0.55,
"output_per_mtok": 2.19,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"as_of": "2026-06",
"_note": "Legacy DeepSeek-R1 pricing. Not on the official current pricing page (now deepseek-v4-flash/pro). Value from LiteLLM deepseek/deepseek-r1."
},
"deepseek-chat": {
"input_per_mtok": 0.14,
"output_per_mtok": 0.28,
"cache_read_per_mtok": 0.0028,
"cache_write_per_mtok": null,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"as_of": "2026-06",
"_note": "Official page (fetched 2026-06): deepseek-chat is deprecated 2026/07/24 and now maps to deepseek-v4-flash non-thinking mode. Live billing = v4-flash: input(cache-miss) 0.14, output 0.28, input(cache-hit) 0.0028."
},
"deepseek-reasoner": {
"input_per_mtok": 0.14,
"output_per_mtok": 0.28,
"cache_read_per_mtok": 0.0028,
"cache_write_per_mtok": null,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"as_of": "2026-06",
"_note": "Official page (fetched 2026-06): deepseek-reasoner is deprecated 2026/07/24 and now maps to deepseek-v4-flash thinking mode. Live billing = v4-flash: input(cache-miss) 0.14, output 0.28, input(cache-hit) 0.0028."
},
"glm-4.6": {
"input_per_mtok": 0.6,
"output_per_mtok": 2.2,
"cache_read_per_mtok": 0.11,
"cache_write_per_mtok": null,
"source": "https://docs.z.ai/guides/overview/pricing",
"as_of": "2026-06",
"_note": "Z.AI international (USD). Cached-input 0.11. Cache-input storage currently limited-time free. Cross-checked vs LiteLLM zai/glm-4.6."
},
"glm-4-plus": {
"input_per_mtok": null,
"output_per_mtok": null,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": null,
"as_of": "2026-06",
"_note": "No authoritative current price found this session. glm-4-plus is a legacy Zhipu/BigModel-native model not listed on the Z.AI international catalog; open.bigmodel.cn pricing is SPA-rendered (no static price) and BigModel English docs returned HTTP 552/404. Not in LiteLLM. Left null per anti-hallucination rule."
},
"glm-4-air": {
"input_per_mtok": 0.2,
"output_per_mtok": 1.1,
"cache_read_per_mtok": 0.03,
"cache_write_per_mtok": null,
"source": "https://docs.z.ai/guides/overview/pricing",
"as_of": "2026-06",
"_note": "Mission billing id glm-4-air; current Z.AI international catalog lists this as GLM-4.5-Air (USD): input 0.2, cached-input 0.03, output 1.1. Cross-checked vs LiteLLM zai/glm-4.5-air (in 0.2/out 1.1)."
}
}
+74
View File
@@ -0,0 +1,74 @@
{
"gemini-2.5-pro": {
"input_per_mtok": 1.25,
"output_per_mtok": 10.0,
"cache_read_per_mtok": 0.125,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "Tiered: prompts >200k tokens cost input $2.50, output $15.00, cache_read $0.25 per Mtok. Cache storage $4.50/Mtok/hr."
},
"gemini-2.5-flash": {
"input_per_mtok": 0.3,
"output_per_mtok": 2.5,
"cache_read_per_mtok": 0.03,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "Text/image/video input; audio input $1.00. Output includes thinking tokens. Cache storage $1.00/Mtok/hr. No >200k context tier."
},
"gemini-2.5-flash-lite": {
"input_per_mtok": 0.1,
"output_per_mtok": 0.4,
"cache_read_per_mtok": 0.01,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "Text/image/video input; audio input $0.30. Cache storage $1.00/Mtok/hr. No >200k context tier."
},
"gemini-2.0-flash": {
"input_per_mtok": 0.1,
"output_per_mtok": 0.4,
"cache_read_per_mtok": 0.025,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "Text/image/video input; audio input $0.70, audio cache_read $0.175. Cache storage $1.00/Mtok/hr. No >200k context tier."
},
"gemini-2.0-flash-lite": {
"input_per_mtok": 0.075,
"output_per_mtok": 0.3,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "DEPRECATED — shut down June 1, 2026. Context caching not available for this model. Prices retained for historical billing."
},
"gemini-2.0-pro": {
"input_per_mtok": null,
"output_per_mtok": null,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "No GA billing id. Only shipped as gemini-2.0-pro-exp (experimental, free of charge); never had paid pricing. Not listed on official pricing page or LiteLLM catalog."
},
"gemini-3-pro-preview": {
"input_per_mtok": 2.0,
"output_per_mtok": 12.0,
"cache_read_per_mtok": 0.2,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "Current 3.x flagship (page labels it Gemini 3.1 Pro Preview). Tiered: prompts >200k cost input $4.00, output $18.00, cache_read $0.40 per Mtok. Cache storage $4.50/Mtok/hr."
},
"gemini-3-flash-preview": {
"input_per_mtok": 0.5,
"output_per_mtok": 3.0,
"cache_read_per_mtok": 0.05,
"cache_write_per_mtok": null,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"as_of": "2026-06",
"_note": "Text/image/video input; audio input $1.00, audio cache_read $0.10. Cache storage $1.00/Mtok/hr. No >200k context tier on standard."
}
}
+106
View File
@@ -0,0 +1,106 @@
{
"gpt-5": {
"input_per_mtok": 1.25,
"output_per_mtok": 10,
"cache_read_per_mtok": 0.125,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-5-mini": {
"input_per_mtok": 0.25,
"output_per_mtok": 2,
"cache_read_per_mtok": 0.025,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-5-nano": {
"input_per_mtok": 0.05,
"output_per_mtok": 0.4,
"cache_read_per_mtok": 0.005,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-5-codex": {
"input_per_mtok": 1.25,
"output_per_mtok": 10,
"cache_read_per_mtok": 0.125,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-4.1": {
"input_per_mtok": 2,
"output_per_mtok": 8,
"cache_read_per_mtok": 0.5,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-4.1-mini": {
"input_per_mtok": 0.4,
"output_per_mtok": 1.6,
"cache_read_per_mtok": 0.1,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-4.1-nano": {
"input_per_mtok": 0.1,
"output_per_mtok": 0.4,
"cache_read_per_mtok": 0.025,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-4o": {
"input_per_mtok": 2.5,
"output_per_mtok": 10,
"cache_read_per_mtok": 1.25,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"gpt-4o-mini": {
"input_per_mtok": 0.15,
"output_per_mtok": 0.6,
"cache_read_per_mtok": 0.075,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"o3": {
"input_per_mtok": 2,
"output_per_mtok": 8,
"cache_read_per_mtok": 0.5,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"o4-mini": {
"input_per_mtok": 1.1,
"output_per_mtok": 4.4,
"cache_read_per_mtok": 0.275,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"o3-mini": {
"input_per_mtok": 1.1,
"output_per_mtok": 4.4,
"cache_read_per_mtok": 0.55,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
},
"codex-mini-latest": {
"input_per_mtok": 1.5,
"output_per_mtok": 6,
"cache_read_per_mtok": 0.375,
"cache_write_per_mtok": null,
"source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json",
"as_of": "2026-06"
}
}
+137
View File
@@ -0,0 +1,137 @@
{
"grok-4": {
"input_per_mtok": 3,
"output_per_mtok": 15,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://docs.x.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Legacy id; xAI public pricing page now lists newer Grok versions (4.3/4.20/build-0.1). Rate verified via LiteLLM xai/ catalog (BerriAI/litellm) which sources directly from xAI."
},
"grok-3": {
"input_per_mtok": 3,
"output_per_mtok": 15,
"cache_read_per_mtok": 0.75,
"cache_write_per_mtok": null,
"source": "https://docs.x.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Legacy id; rate from LiteLLM xai/grok-3 catalog (sourced from xAI)."
},
"grok-code-fast-1": {
"input_per_mtok": 0.2,
"output_per_mtok": 1.5,
"cache_read_per_mtok": 0.02,
"cache_write_per_mtok": null,
"source": "https://docs.x.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Legacy id; rate from LiteLLM xai/grok-code-fast-1 catalog (sourced from xAI)."
},
"grok-2": {
"input_per_mtok": 2,
"output_per_mtok": 10,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://docs.x.ai/docs/pricing",
"as_of": "2026-06",
"_note": "Legacy id (grok-2-1212); rate from LiteLLM xai/grok-2 catalog (sourced from xAI)."
},
"mistral-large-latest": {
"input_per_mtok": 0.5,
"output_per_mtok": 1.5,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://mistral.ai/pricing",
"as_of": "2026-06",
"_note": "mistral-large-latest = Mistral Large 3."
},
"codestral-latest": {
"input_per_mtok": 0.3,
"output_per_mtok": 0.9,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://mistral.ai/pricing",
"as_of": "2026-06",
"_note": "Codestral 2; price cross-confirmed via Google Vertex partner pricing ($0.30/$0.90)."
},
"devstral": {
"input_per_mtok": 0.4,
"output_per_mtok": 2,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://mistral.ai/pricing",
"as_of": "2026-06",
"_note": "Maps to devstral-medium-latest (Mistral hosted API). devstral-small is $0.10/$0.30."
},
"mistral-medium": {
"input_per_mtok": 0.4,
"output_per_mtok": 2,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://mistral.ai/pricing",
"as_of": "2026-06",
"_note": "Mistral Medium 3; cross-confirmed via Google Vertex partner pricing ($0.40/$2.00)."
},
"llama-4-maverick": {
"input_per_mtok": 0.27,
"output_per_mtok": 0.85,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.together.ai/pricing",
"as_of": "2026-06",
"_note": "Host-dependent. Priced via Together AI serverless. Other hosts differ (e.g. AWS Bedrock, Vertex $0.35/$1.15)."
},
"llama-4-scout": {
"input_per_mtok": 0.08,
"output_per_mtok": 0.3,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.together.ai/pricing",
"as_of": "2026-06",
"_note": "Host-dependent. Priced via Together AI serverless. Vertex lists $0.25/$0.70."
},
"llama-3.3-70b": {
"input_per_mtok": 0.88,
"output_per_mtok": 0.88,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://www.together.ai/pricing",
"as_of": "2026-06",
"_note": "Host-dependent. Together Llama-3.3-70B-Instruct-Turbo ($0.88/$0.88, from LiteLLM together_ai source). Cheaper on DeepInfra ($0.13/$0.39); Vertex $0.72/$0.72."
},
"command-a": {
"input_per_mtok": 2.5,
"output_per_mtok": 10,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://cohere.com/pricing",
"as_of": "2026-06",
"_note": "command-a-03-2025."
},
"command-r-plus": {
"input_per_mtok": 2.5,
"output_per_mtok": 10,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://cohere.com/pricing",
"as_of": "2026-06",
"_note": "command-r-plus-08-2024 (legacy)."
},
"amazon-nova-pro": {
"input_per_mtok": 0.8,
"output_per_mtok": 3.2,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://aws.amazon.com/bedrock/pricing/",
"as_of": "2026-06",
"_note": "AWS Bedrock on-demand ($0.0008/$0.0032 per 1K tokens)."
},
"amazon-nova-lite": {
"input_per_mtok": 0.06,
"output_per_mtok": 0.24,
"cache_read_per_mtok": null,
"cache_write_per_mtok": null,
"source": "https://aws.amazon.com/bedrock/pricing/",
"as_of": "2026-06",
"_note": "AWS Bedrock on-demand ($0.00006/$0.00024 per 1K tokens)."
}
}
+81 -44
View File
@@ -5,6 +5,11 @@
* All 13 event categories as specified in PRD Section 3.
*/
import {
lookupPrice as catalogLookupPrice,
computeCostUsd as catalogComputeCostUsd,
} from "../pricing/catalog.js";
// ── Public interfaces ──────────────────────────────────────────────────────
export interface SessionEvent {
@@ -1339,62 +1344,92 @@ function extractFileReadMetadata(input: HookInput): SessionEvent[] {
}
/**
* Per-model USD price table — Anthropic public list pricing, $/MTok.
* Verified against platform.claude.com/docs/en/about-claude/pricing,
* cloudzero.com, finout.io 2026-06 (cache: 5-min cache_write = 1.25× input,
* cache_read = 0.10× input). Fast-mode variants (e.g. opus-4-8-fast at
* $10/$50) are intentionally NOT mapped — they ship as separate model
* ids and would dilute the standard-tier dashboards if blended here.
* Per-model USD pricing now lives in the curated multi-vendor catalog
* (src/pricing/catalog.ts), which prices each model from ITS OWN row across
* Anthropic / OpenAI / Google / Chinese / other vendors. This kills the old
* bug where the hardcoded Anthropic-only table here billed every non-Claude
* model at Claude-Sonnet's `default` rate. Unknown ids now resolve to a null
* cost (one console.warn) instead of a silently wrong Claude rate.
*
* NOTE: 16-oss-verify-gap-prd Gap #1 quoted Opus at $15/$75 — that is
* the prior Opus 4 (non-4.7) rate. Opus 4.7 and 4.8 ship at $5/$25.
* resolveModelId picks the first non-empty model id from the hook candidates;
* date-suffixed ids (e.g. claude-haiku-4-5-20251001) are reduced to a catalog
* hit by progressively dropping trailing `-segment` suffixes (NO regex).
*/
const MODEL_PRICING_USD_PER_MTOK: Record<string, {
input: number;
output: number;
cache_write: number;
cache_read: number;
}> = {
"claude-opus-4-8": { input: 5.00, output: 25.00, cache_write: 6.25, cache_read: 0.50 },
"claude-opus-4-7": { input: 5.00, output: 25.00, cache_write: 6.25, cache_read: 0.50 },
"claude-sonnet-4-6": { input: 3.00, output: 15.00, cache_write: 3.75, cache_read: 0.30 },
"claude-haiku-4-5": { input: 1.00, output: 5.00, cache_write: 1.25, cache_read: 0.10 },
default: { input: 3.00, output: 15.00, cache_write: 3.75, cache_read: 0.30 },
};
function resolveModelKey(input: HookInput, parsedResp: Record<string, unknown>): string {
function resolveModelId(input: HookInput, parsedResp: Record<string, unknown>): string {
const candidates: unknown[] = [
input.tool_input?.model,
(input as unknown as Record<string, unknown>).model,
parsedResp.model,
];
const keys = Object.keys(MODEL_PRICING_USD_PER_MTOK).filter((k) => k !== "default");
for (const c of candidates) {
if (typeof c !== "string" || c.length === 0) continue;
if (c in MODEL_PRICING_USD_PER_MTOK) return c;
// Prefix match for date-suffixed model ids
// (e.g. claude-haiku-4-5-20251001 → claude-haiku-4-5)
for (const key of keys) {
if (c.startsWith(key)) return key;
}
if (typeof c === "string" && c.length > 0) return c;
}
return "default";
return "";
}
function computeCostUsd(
modelKey: string,
/**
* Drop one trailing `-<segment>` from a model id, char-algorithmically (no
* regex): walks back to the last '-' and returns the head, or null when there
* is no usable separator. Lets a date-suffixed id fall back to its base id
* (claude-haiku-4-5-20251001 → claude-haiku-4-5 → … ) one segment at a time.
*/
function dropTrailingSegment(id: string): string | null {
for (let i = id.length - 1; i > 0; i--) {
if (id.charCodeAt(i) === 45 /* '-' */) return id.slice(0, i);
}
return null;
}
/**
* Resolve a model id to one the catalog can price: try the raw id, then
* progressively trim trailing `-segment` suffixes so a date-suffixed id still
* prices off its base model. Probes with lookupPrice (no warn) and returns the
* first id that hits, or "" on a full miss — so cost compute warns at most once.
*/
function resolveCatalogId(modelId: string): string {
let candidate: string | null = modelId;
while (candidate && candidate.length > 0) {
if (catalogLookupPrice(candidate) !== null) return candidate;
candidate = dropTrailingSegment(candidate);
}
return "";
}
/**
* Cost for a turn via the catalog. Returns null on a price miss (catalog emits
* one console.warn of the unmatched id) or when all token buckets are zero.
*/
function computeTurnCostUsd(
modelId: string,
inputTokens: number,
outputTokens: number,
cacheCreationTokens: number,
cacheReadTokens: number,
): number {
const price = MODEL_PRICING_USD_PER_MTOK[modelKey] ?? MODEL_PRICING_USD_PER_MTOK.default;
const totalMicroDollars =
inputTokens * price.input +
outputTokens * price.output +
cacheCreationTokens * price.cache_write +
cacheReadTokens * price.cache_read;
return totalMicroDollars / 1_000_000;
): number | null {
const resolved = resolveCatalogId(modelId);
// Feed the resolved id when found; otherwise pass the raw id so the catalog's
// single miss-warning carries the id the operator actually saw.
return catalogComputeCostUsd(resolved || modelId, {
input_tokens: inputTokens,
output_tokens: outputTokens,
cache_creation_tokens: cacheCreationTokens,
cache_read_tokens: cacheReadTokens,
});
}
/**
* Format a cost to a compact `cost_usd` string, char-algorithmically (no
* regex). Renders 6 decimals, drops trailing zeros, and keeps a single `.0`
* when the fraction trims to empty (e.g. 0 → "0.0"), matching the prior
* `.toFixed(6).replace(...)` output exactly.
*/
function formatCostUsd(cost: number): string {
let s = cost.toFixed(6);
let end = s.length;
while (end > 0 && s.charCodeAt(end - 1) === 48 /* '0' */) end--;
s = s.slice(0, end);
if (s.length > 0 && s.charCodeAt(s.length - 1) === 46 /* '.' */) s += "0";
return s;
}
/**
@@ -1454,9 +1489,11 @@ function extractAgentUsage(input: HookInput): SessionEvent[] {
: 0;
const anyTokens = inputTokens > 0 || outputTokens > 0 || cacheCreate > 0 || cacheRead > 0;
if (anyTokens) {
const modelKey = resolveModelKey(input, out);
const cost = computeCostUsd(modelKey, inputTokens, outputTokens, cacheCreate, cacheRead);
parts.push(`cost_usd:${cost.toFixed(6).replace(/0+$/, "").replace(/\.$/, ".0")}`);
const modelId = resolveModelId(input, out);
// null ⇒ unmatched model id (catalog warned once) — skip the cost token
// rather than blend a wrong Claude rate (the old non-Claude bug).
const cost = computeTurnCostUsd(modelId, inputTokens, outputTokens, cacheCreate, cacheRead);
if (cost !== null) parts.push(`cost_usd:${formatCostUsd(cost)}`);
}
return [{
+177
View File
@@ -0,0 +1,177 @@
/**
* Pricing catalog — curated multi-vendor price lookup + cost compute.
*
* Replaces the Anthropic-only hardcoded table that lived in
* src/session/extract.ts (≈1356-1396). That table priced EVERY model at
* a Claude rate (non-Claude models silently inherited Sonnet's default),
* which over-/under-charged every OpenAI / Gemini / Qwen / DeepSeek turn.
*
* The catalog merges the 5 curated vendor JSONs (src/pricing/sources/*.json)
* into one per-Mtok price map and prices each model from ITS OWN row.
* Behaviours under test:
* (a) curated lookup hits across all five vendors
* (b) provider/ prefix strip (anthropic/claude-opus-4-8 → claude-opus-4-8)
* (c) non-Claude model gets ITS price, not Claude's (the bug)
* (d) unknown id → null + one console.warn of the unmatched id
* (e) cache_read vs cache_creation weighting (distinct prices)
* (f) all-zero / absent tokens → null
* (g) native-cost passthrough wins over computed
*/
import { describe, test, expect, vi } from "vitest";
import {
lookupPrice,
computeCostUsd,
nativeOrComputed,
} from "../../src/pricing/catalog.js";
describe("pricing/catalog — lookupPrice", () => {
// (a) curated lookup hits
test("(a) Anthropic curated hit returns per-Mtok price", () => {
const p = lookupPrice("claude-opus-4-8");
expect(p).not.toBeNull();
expect(p!.input_per_mtok).toBe(5);
expect(p!.output_per_mtok).toBe(25);
expect(p!.cache_write_per_mtok).toBe(6.25);
expect(p!.cache_read_per_mtok).toBe(0.5);
});
test("(a) OpenAI curated hit", () => {
expect(lookupPrice("gpt-5")!.input_per_mtok).toBe(1.25);
});
test("(a) Google curated hit", () => {
expect(lookupPrice("gemini-2.5-flash")!.output_per_mtok).toBe(2.5);
});
test("(a) Chinese vendor curated hit", () => {
expect(lookupPrice("deepseek-v3")).not.toBeNull();
});
test("(a) Other vendor curated hit", () => {
expect(lookupPrice("grok-4")).not.toBeNull();
});
// (b) provider/ prefix strip + normalization
test("(b) provider/ prefix is stripped", () => {
const stripped = lookupPrice("anthropic/claude-opus-4-8");
const bare = lookupPrice("claude-opus-4-8");
expect(stripped).toEqual(bare);
});
test("(b) normalization trims and lowercases", () => {
expect(lookupPrice(" GPT-5 ")).toEqual(lookupPrice("gpt-5"));
});
test("(b) openrouter-style provider/model strips first segment only", () => {
// openai/gpt-5 → gpt-5
expect(lookupPrice("openai/gpt-5")).toEqual(lookupPrice("gpt-5"));
});
// (d) unknown id → null
test("(d) unknown model id returns null", () => {
expect(lookupPrice("totally-made-up-model-xyz")).toBeNull();
});
// null-priced curated entry → treated as no price
test("null-priced curated entry (gemini-2.0-pro) is treated as no price", () => {
expect(lookupPrice("gemini-2.0-pro")).toBeNull();
});
});
describe("pricing/catalog — computeCostUsd", () => {
// (c) THE BUG: non-Claude model must use its own price, not Claude's.
test("(c) gpt-5 priced from its own row, NOT Claude default", () => {
const tokens = { input_tokens: 1000, output_tokens: 500 };
const gpt5 = computeCostUsd("gpt-5", tokens);
// gpt-5: 1000*1.25 + 500*10 = 1250 + 5000 = 6250 / 1e6 = 0.00625
expect(gpt5).toBeCloseTo(0.00625, 8);
// Claude-default (old buggy path) would have produced Sonnet 0.0105.
const sonnet = computeCostUsd("claude-sonnet-4-6", tokens);
expect(gpt5).not.toBeCloseTo(sonnet!, 8);
});
test("(c) gemini-2.5-flash priced from its own row", () => {
// 1000*0.3 + 500*2.5 = 300 + 1250 = 1550 / 1e6 = 0.00155
expect(computeCostUsd("gemini-2.5-flash", {
input_tokens: 1000,
output_tokens: 500,
})).toBeCloseTo(0.00155, 8);
});
// opus-4-8 regression — must match the old hardcoded table exactly.
test("opus-4-8 regression: 1000 in + 500 out = 0.0175", () => {
expect(computeCostUsd("claude-opus-4-8", {
input_tokens: 1000,
output_tokens: 500,
})).toBeCloseTo(0.0175, 8);
});
// (e) cache_read vs cache_creation weighting
test("(e) cache_creation uses cache_write price, cache_read uses cache_read price", () => {
// sonnet: in 3, out 15, cache_write 3.75, cache_read 0.30
// 1000*3 + 500*15 + 1000*3.75 + 1500*0.30 = 3000+7500+3750+450 = 14700
expect(computeCostUsd("claude-sonnet-4-6", {
input_tokens: 1000,
output_tokens: 500,
cache_creation_tokens: 1000,
cache_read_tokens: 1500,
})).toBeCloseTo(0.0147, 8);
});
test("(e) null cache price falls back to input price for that bucket", () => {
// gpt-5 has cache_write_per_mtok=null → cache_creation billed at input (1.25).
// 2000 cache_creation tokens only: 2000*1.25 = 2500 / 1e6 = 0.0025
expect(computeCostUsd("gpt-5", {
cache_creation_tokens: 2000,
})).toBeCloseTo(0.0025, 8);
});
// (f) all-zero / absent tokens → null
test("(f) all-zero tokens returns null", () => {
expect(computeCostUsd("claude-sonnet-4-6", {
input_tokens: 0,
output_tokens: 0,
cache_read_tokens: 0,
cache_creation_tokens: 0,
})).toBeNull();
});
test("(f) absent tokens returns null", () => {
expect(computeCostUsd("claude-sonnet-4-6", {})).toBeNull();
});
// (d) unknown model → null + single console.warn of the unmatched id
test("(d) unknown model returns null and warns once with the id", () => {
const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
try {
const cost = computeCostUsd("unknown-vendor/mystery-9000", {
input_tokens: 1000,
});
expect(cost).toBeNull();
expect(warn).toHaveBeenCalledTimes(1);
expect(String(warn.mock.calls[0]?.[0])).toContain("mystery-9000");
} finally {
warn.mockRestore();
}
});
});
describe("pricing/catalog — nativeOrComputed", () => {
// (g) native-cost passthrough
test("(g) provider native cost wins over computed", () => {
const native = nativeOrComputed("gpt-5", { input_tokens: 1000 }, 0.42);
expect(native).toBe(0.42);
});
test("(g) falls back to computed when native is null/undefined", () => {
const computed = computeCostUsd("gpt-5", { input_tokens: 1000 });
expect(nativeOrComputed("gpt-5", { input_tokens: 1000 }, null)).toBe(computed);
expect(nativeOrComputed("gpt-5", { input_tokens: 1000 })).toBe(computed);
});
test("(g) native cost of 0 is a real value, not a miss", () => {
// A provider that genuinely charged $0 (free tier) must pass through.
expect(nativeOrComputed("gpt-5", { input_tokens: 1000 }, 0)).toBe(0);
});
});
+17 -5
View File
@@ -19,7 +19,7 @@
* forward-compatible Zod envelope accepts them today (no migration).
*/
import { describe, test, expect } from "vitest";
import { describe, test, expect, vi } from "vitest";
import { extractEvents } from "../../src/session/extract.js";
function agentUsageOf(toolResponse: unknown, toolName: string = "Task") {
@@ -211,7 +211,12 @@ describe("extractAgentUsage — Issue #4 AgentOutput.usage capture", () => {
expect(events[0].data).toMatch(/cost_usd:0\.0147/);
});
test("cost_usd: unknown model falls back to default Sonnet pricing", () => {
test("cost_usd: unknown model → no cost emitted (was: wrong Claude fallback)", () => {
// Catalog rewire — an unmatched model id no longer inherits Claude-Sonnet's
// rate (the old `default` bug). It resolves to null cost, so the event is
// still emitted (token counts) but carries NO cost_usd token. The catalog
// also console.warns the unmatched id once; silence it here.
const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
const events = extractEvents({
tool_name: "Task",
tool_input: { model: "claude-future-model-99" },
@@ -220,7 +225,9 @@ describe("extractAgentUsage — Issue #4 AgentOutput.usage capture", () => {
usage: { input_tokens: 1000, output_tokens: 500 },
}),
}).filter((e) => e.type === "agent_usage");
expect(events[0].data).toMatch(/cost_usd:0\.0105/);
warn.mockRestore();
expect(events.length).toBe(1);
expect(events[0].data).not.toMatch(/cost_usd:/);
});
test("cost_usd: Opus 4.8 priced at same standard rate as Opus 4.7", () => {
@@ -255,7 +262,10 @@ describe("extractAgentUsage — Issue #4 AgentOutput.usage capture", () => {
expect(events[0].data).toMatch(/cost_usd:0\.0035/);
});
test("cost_usd: no model + token counts → still computes with default pricing", () => {
test("cost_usd: no model id → no cost emitted (cannot price without a model)", () => {
// Catalog rewire — without a model id there is no row to price from, so no
// cost_usd is blended. The empty id is not warned (it is not a real miss).
const warn = vi.spyOn(console, "warn").mockImplementation(() => {});
const events = extractEvents({
tool_name: "Task",
tool_input: {},
@@ -263,7 +273,9 @@ describe("extractAgentUsage — Issue #4 AgentOutput.usage capture", () => {
usage: { input_tokens: 1000, output_tokens: 500 },
}),
}).filter((e) => e.type === "agent_usage");
expect(events[0].data).toMatch(/cost_usd:/);
warn.mockRestore();
expect(events.length).toBe(1);
expect(events[0].data).not.toMatch(/cost_usd:/);
});
test("cost_usd: zero tokens → cost_usd:0 not emitted (skip)", () => {
+80
View File
@@ -0,0 +1,80 @@
# LiteLLM Catalog — Adapter Notes
Vendored from BerriAI/litellm as context-mode's comprehensive pricing base, so
unknown/unseen models still resolve a price instead of failing.
- **Source:** `https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json`
- **Local file:** `litellm-catalog.json`
- **Size:** ~1.53 MB (1,570,159 bytes)
- **Top-level keys:** 2,901 (2,900 model entries + 1 `sample_spec` template — skip `sample_spec`)
- **Entries with numeric `input_cost_per_token`:** 2,432
- **Distinct `litellm_provider` values:** 120 (top: fireworks_ai, bedrock, openai, azure, gemini, mistral, openrouter …)
## Shape
The file is a flat JSON object. **Each key is a model id**; each value is a metadata object.
There is no wrapper array. One special key, `sample_spec`, is a documentation template
(not a real model) and must be excluded.
```jsonc
{
"sample_spec": { /* template — ignore */ },
"gpt-4o": { "input_cost_per_token": 0.0000025, ... },
"claude-sonnet-4-20250514": { ... }
}
```
## Cost fields — verified present in the fetched JSON
> All `*_cost_per_token` values are **PER TOKEN** (USD), not per-million.
> To convert to per-Mtok (context-mode's internal unit): **multiply by `1e6` (×1,000,000).**
> e.g. `gpt-4o` `input_cost_per_token = 0.0000025` → `0.0000025 × 1e6 = $2.50 / Mtok`.
Core fields (use these; coverage count in parentheses):
| Field | Meaning | Count |
|-------|---------|-------|
| `input_cost_per_token` | prompt cost per token | 2,433 |
| `output_cost_per_token` | completion cost per token | 2,430 |
| `cache_read_input_token_cost` | cached-prompt read cost per token | 645 |
| `cache_creation_input_token_cost` | cache-write/creation cost per token | 210 |
Real example (`claude-sonnet-4-20250514`, provider `anthropic`):
`input_cost_per_token: 0.000003` (= $3.00/Mtok), `output_cost_per_token: 0.000015` (= $15.00/Mtok),
`cache_read_input_token_cost: 3e-7` (= $0.30/Mtok), `cache_creation_input_token_cost: 0.00000375` (= $3.75/Mtok).
### Schema surprises / gotchas
- **Tiered & variant cost fields exist** — e.g. `input_cost_per_token_above_200k_tokens`,
`cache_read_input_token_cost_above_200k_tokens`, `_priority`, `_batches`, `_flex`,
`_above_1hr`, `_above_272k_tokens`. Treat these as optional overrides; fall back to the
base `input_cost_per_token` / `output_cost_per_token`.
- **Non-token cost units also appear** and are NOT per-token: `input_cost_per_second`,
`input_cost_per_character`, `input_cost_per_image`, `input_cost_per_pixel`,
`output_cost_per_second`, `output_cost_per_reasoning_token`,
`search_context_cost_per_query` (the last can be an object keyed by search-context size).
Do not blindly ×1e6 these — only the `*_per_token` family is per-token.
- **2,366 of 2,900 keys contain `/`** — namespaced ids like `bedrock/...`,
`1024-x-1024/dall-e-2`, image/resolution-prefixed entries. Match on the full key.
- Metadata fields used for context limits: `max_tokens`, `max_input_tokens`,
`max_output_tokens` (and `mode` distinguishes `chat`, `embedding`, image, etc.).
- Some entries carry `deprecation_date` (81 entries).
## Lookup strategy (for the adapter)
Given an incoming `model_id`:
1. **Exact match** — `catalog[model_id]`. Fastest; covers the common case.
2. **Provider-stripped / namespaced fallback** — many ids are `provider/model`.
Try stripping or adding a known provider prefix:
- if `model_id` has no `/`, try `catalog[provider + "/" + model_id]`;
- if `model_id` is `provider/model`, also try the bare `model` segment.
3. **Provider + model match** — scan entries whose `litellm_provider` matches the resolved
provider and whose key endsWith the model segment.
4. **Skip `sample_spec`** in every scan.
5. From the matched entry, read `input_cost_per_token` / `output_cost_per_token`
(+ optional `cache_read_input_token_cost`, `cache_creation_input_token_cost`),
then **× 1e6** to get per-Mtok rates.
6. If nothing matches or `input_cost_per_token` is absent (some entries price only by
second/character/image), the model has no usable per-token price — fall through to
context-mode's own default.