From aaf7916ffff9e326b797cb5bf46b100036304327 Mon Sep 17 00:00:00 2001 From: Mert Koseoglu Date: Fri, 26 Jun 2026 00:46:53 +0300 Subject: [PATCH] refactor(pricing): single catalog under src/session; structured cost event Consolidates the 5 vendor price files into one src/session/model-prices.json (61 models), moves the module to src/session/pricing.ts beside its only consumer (extract.ts), and drops the dev-only litellm NOTES from tracking (litellm base stays gitignored). Wave 2b: extractAgentUsage emits model_id + input/output/cache_read/cache_creation tokens + cost_usd as structured top-level event fields, replacing the colon-string, so the platform persists them as columns. 45 tests green, tsc clean. --- .gitignore | 3 +- src/pricing/sources/anthropic.json | 82 ---- src/pricing/sources/chinese.json | 155 ------- src/pricing/sources/google.json | 74 --- src/pricing/sources/openai.json | 106 ----- src/pricing/sources/others.json | 137 ------ src/session/extract.ts | 41 +- src/session/model-prices.json | 429 ++++++++++++++++++ .../catalog.ts => session/pricing.ts} | 62 +-- tests/session/extract-agent-usage.test.ts | 87 ++++ .../pricing.test.ts} | 10 +- tools/pricing/litellm-NOTES.md | 80 ---- 12 files changed, 583 insertions(+), 683 deletions(-) delete mode 100644 src/pricing/sources/anthropic.json delete mode 100644 src/pricing/sources/chinese.json delete mode 100644 src/pricing/sources/google.json delete mode 100644 src/pricing/sources/openai.json delete mode 100644 src/pricing/sources/others.json create mode 100644 src/session/model-prices.json rename src/{pricing/catalog.ts => session/pricing.ts} (75%) rename tests/{pricing/catalog.test.ts => session/pricing.test.ts} (95%) delete mode 100644 tools/pricing/litellm-NOTES.md diff --git a/.gitignore b/.gitignore index 2e2ea416..b1c6453d 100644 --- a/.gitignore +++ b/.gitignore @@ -40,8 +40,7 @@ context-mode-guidance-*/ .vibetree/ .cw/ # Dev-only pricing source: the full litellm catalog (~1.5MB) is the manual -# refresh base for src/pricing/sources/*.json — never bundled into the plugin. -# Refresh: see tools/pricing/litellm-NOTES.md. Keep NOTES tracked, ignore the blob. +# refresh base for src/session/model-prices.json — never bundled into the plugin. tools/pricing/litellm-catalog.json /.cocoindex_code/ /.kilo/ diff --git a/src/pricing/sources/anthropic.json b/src/pricing/sources/anthropic.json deleted file mode 100644 index a15cad69..00000000 --- a/src/pricing/sources/anthropic.json +++ /dev/null @@ -1,82 +0,0 @@ -{ - "claude-opus-4-8": { - "input_per_mtok": 5.00, - "output_per_mtok": 25.00, - "cache_read_per_mtok": 0.50, - "cache_write_per_mtok": 6.25, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-opus-4-7": { - "input_per_mtok": 5.00, - "output_per_mtok": 25.00, - "cache_read_per_mtok": 0.50, - "cache_write_per_mtok": 6.25, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-opus-4-6": { - "input_per_mtok": 5.00, - "output_per_mtok": 25.00, - "cache_read_per_mtok": 0.50, - "cache_write_per_mtok": 6.25, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-opus-4-5": { - "input_per_mtok": 5.00, - "output_per_mtok": 25.00, - "cache_read_per_mtok": 0.50, - "cache_write_per_mtok": 6.25, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-sonnet-4-6": { - "input_per_mtok": 3.00, - "output_per_mtok": 15.00, - "cache_read_per_mtok": 0.30, - "cache_write_per_mtok": 3.75, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-sonnet-4-5": { - "input_per_mtok": 3.00, - "output_per_mtok": 15.00, - "cache_read_per_mtok": 0.30, - "cache_write_per_mtok": 3.75, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-haiku-4-5": { - "input_per_mtok": 1.00, - "output_per_mtok": 5.00, - "cache_read_per_mtok": 0.10, - "cache_write_per_mtok": 1.25, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-3-7-sonnet": { - "input_per_mtok": 3.00, - "output_per_mtok": 15.00, - "cache_read_per_mtok": 0.30, - "cache_write_per_mtok": 3.75, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "claude-3-5-haiku": { - "input_per_mtok": 0.80, - "output_per_mtok": 4.00, - "cache_read_per_mtok": 0.08, - "cache_write_per_mtok": 1.00, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - }, - "claude-fable-5": { - "input_per_mtok": 10.00, - "output_per_mtok": 50.00, - "cache_read_per_mtok": 1.00, - "cache_write_per_mtok": 12.50, - "source": "https://platform.claude.com/docs/en/about-claude/pricing", - "as_of": "2026-06" - } -} diff --git a/src/pricing/sources/chinese.json b/src/pricing/sources/chinese.json deleted file mode 100644 index 15e78b09..00000000 --- a/src/pricing/sources/chinese.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "qwen3-coder": { - "input_per_mtok": 1.0, - "output_per_mtok": 5.0, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.alibabacloud.com/help/en/model-studio/models", - "as_of": "2026-06", - "_note": "Billing id maps to qwen3-coder-plus. Tiered by input length; headline is the 0-32K tier. Higher tiers: 32K-128K in 1.8/out 9, 128K-256K in 3/out 15, 256K-1M in 6/out 60. Cross-checked vs LiteLLM tiered_pricing." - }, - "qwen-max": { - "input_per_mtok": 1.6, - "output_per_mtok": 6.4, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.alibabacloud.com/help/en/model-studio/models", - "as_of": "2026-06", - "_note": "Alibaba DashScope international (USD-denominated). Cross-checked vs LiteLLM dashscope/qwen-max." - }, - "qwen-plus": { - "input_per_mtok": 0.4, - "output_per_mtok": 1.2, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.alibabacloud.com/help/en/model-studio/models", - "as_of": "2026-06", - "_note": "Alibaba DashScope international (USD). Cross-checked vs LiteLLM dashscope/qwen-plus." - }, - "qwen-turbo": { - "input_per_mtok": 0.05, - "output_per_mtok": 0.2, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.alibabacloud.com/help/en/model-studio/models", - "as_of": "2026-06", - "_note": "Alibaba DashScope international (USD). Cross-checked vs LiteLLM dashscope/qwen-turbo." - }, - "qwen3-max": { - "input_per_mtok": 1.2, - "output_per_mtok": 6.0, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.alibabacloud.com/help/en/model-studio/models", - "as_of": "2026-06", - "_note": "Tiered by input length; headline is the 0-32K tier. Higher tiers: 32K-128K in 2.4/out 12, 128K-252K in 3/out 15. Cross-checked vs LiteLLM dashscope/qwen3-max tiered_pricing." - }, - "kimi-k2": { - "input_per_mtok": 0.6, - "output_per_mtok": 2.5, - "cache_read_per_mtok": 0.15, - "cache_write_per_mtok": null, - "source": "https://platform.moonshot.ai/docs/pricing/chat", - "as_of": "2026-06", - "_note": "Moonshot international platform (USD). Resolves to kimi-k2-0905-preview. Cache-hit input 0.15. Cross-checked vs LiteLLM moonshot/kimi-k2-0905-preview." - }, - "kimi-k2-turbo": { - "input_per_mtok": 1.15, - "output_per_mtok": 8.0, - "cache_read_per_mtok": 0.15, - "cache_write_per_mtok": null, - "source": "https://platform.moonshot.ai/docs/pricing/chat", - "as_of": "2026-06", - "_note": "Moonshot international (USD). Resolves to kimi-k2-turbo-preview. Cache-hit input 0.15. Cross-checked vs LiteLLM moonshot/kimi-k2-turbo-preview." - }, - "moonshot-v1-8k": { - "input_per_mtok": 0.2, - "output_per_mtok": 2.0, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://platform.moonshot.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Moonshot international (USD). Cross-checked vs LiteLLM moonshot/moonshot-v1-8k." - }, - "moonshot-v1-32k": { - "input_per_mtok": 1.0, - "output_per_mtok": 3.0, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://platform.moonshot.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Moonshot international (USD). Cross-checked vs LiteLLM moonshot/moonshot-v1-32k." - }, - "moonshot-v1-128k": { - "input_per_mtok": 2.0, - "output_per_mtok": 5.0, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://platform.moonshot.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Moonshot international (USD). Cross-checked vs LiteLLM moonshot/moonshot-v1-128k." - }, - "deepseek-v3": { - "input_per_mtok": 0.27, - "output_per_mtok": 1.1, - "cache_read_per_mtok": 0.07, - "cache_write_per_mtok": 0.0, - "source": "https://api-docs.deepseek.com/quick_start/pricing", - "as_of": "2026-06", - "_note": "Legacy DeepSeek-V3 era pricing. No longer shown on the official current pricing page (which now lists deepseek-v4-flash/pro). Value from LiteLLM deepseek/deepseek-v3, last sourced from the official page. cache-hit input 0.07." - }, - "deepseek-r1": { - "input_per_mtok": 0.55, - "output_per_mtok": 2.19, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://api-docs.deepseek.com/quick_start/pricing", - "as_of": "2026-06", - "_note": "Legacy DeepSeek-R1 pricing. Not on the official current pricing page (now deepseek-v4-flash/pro). Value from LiteLLM deepseek/deepseek-r1." - }, - "deepseek-chat": { - "input_per_mtok": 0.14, - "output_per_mtok": 0.28, - "cache_read_per_mtok": 0.0028, - "cache_write_per_mtok": null, - "source": "https://api-docs.deepseek.com/quick_start/pricing", - "as_of": "2026-06", - "_note": "Official page (fetched 2026-06): deepseek-chat is deprecated 2026/07/24 and now maps to deepseek-v4-flash non-thinking mode. Live billing = v4-flash: input(cache-miss) 0.14, output 0.28, input(cache-hit) 0.0028." - }, - "deepseek-reasoner": { - "input_per_mtok": 0.14, - "output_per_mtok": 0.28, - "cache_read_per_mtok": 0.0028, - "cache_write_per_mtok": null, - "source": "https://api-docs.deepseek.com/quick_start/pricing", - "as_of": "2026-06", - "_note": "Official page (fetched 2026-06): deepseek-reasoner is deprecated 2026/07/24 and now maps to deepseek-v4-flash thinking mode. Live billing = v4-flash: input(cache-miss) 0.14, output 0.28, input(cache-hit) 0.0028." - }, - "glm-4.6": { - "input_per_mtok": 0.6, - "output_per_mtok": 2.2, - "cache_read_per_mtok": 0.11, - "cache_write_per_mtok": null, - "source": "https://docs.z.ai/guides/overview/pricing", - "as_of": "2026-06", - "_note": "Z.AI international (USD). Cached-input 0.11. Cache-input storage currently limited-time free. Cross-checked vs LiteLLM zai/glm-4.6." - }, - "glm-4-plus": { - "input_per_mtok": null, - "output_per_mtok": null, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": null, - "as_of": "2026-06", - "_note": "No authoritative current price found this session. glm-4-plus is a legacy Zhipu/BigModel-native model not listed on the Z.AI international catalog; open.bigmodel.cn pricing is SPA-rendered (no static price) and BigModel English docs returned HTTP 552/404. Not in LiteLLM. Left null per anti-hallucination rule." - }, - "glm-4-air": { - "input_per_mtok": 0.2, - "output_per_mtok": 1.1, - "cache_read_per_mtok": 0.03, - "cache_write_per_mtok": null, - "source": "https://docs.z.ai/guides/overview/pricing", - "as_of": "2026-06", - "_note": "Mission billing id glm-4-air; current Z.AI international catalog lists this as GLM-4.5-Air (USD): input 0.2, cached-input 0.03, output 1.1. Cross-checked vs LiteLLM zai/glm-4.5-air (in 0.2/out 1.1)." - } -} diff --git a/src/pricing/sources/google.json b/src/pricing/sources/google.json deleted file mode 100644 index 09399361..00000000 --- a/src/pricing/sources/google.json +++ /dev/null @@ -1,74 +0,0 @@ -{ - "gemini-2.5-pro": { - "input_per_mtok": 1.25, - "output_per_mtok": 10.0, - "cache_read_per_mtok": 0.125, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "Tiered: prompts >200k tokens cost input $2.50, output $15.00, cache_read $0.25 per Mtok. Cache storage $4.50/Mtok/hr." - }, - "gemini-2.5-flash": { - "input_per_mtok": 0.3, - "output_per_mtok": 2.5, - "cache_read_per_mtok": 0.03, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "Text/image/video input; audio input $1.00. Output includes thinking tokens. Cache storage $1.00/Mtok/hr. No >200k context tier." - }, - "gemini-2.5-flash-lite": { - "input_per_mtok": 0.1, - "output_per_mtok": 0.4, - "cache_read_per_mtok": 0.01, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "Text/image/video input; audio input $0.30. Cache storage $1.00/Mtok/hr. No >200k context tier." - }, - "gemini-2.0-flash": { - "input_per_mtok": 0.1, - "output_per_mtok": 0.4, - "cache_read_per_mtok": 0.025, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "Text/image/video input; audio input $0.70, audio cache_read $0.175. Cache storage $1.00/Mtok/hr. No >200k context tier." - }, - "gemini-2.0-flash-lite": { - "input_per_mtok": 0.075, - "output_per_mtok": 0.3, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "DEPRECATED — shut down June 1, 2026. Context caching not available for this model. Prices retained for historical billing." - }, - "gemini-2.0-pro": { - "input_per_mtok": null, - "output_per_mtok": null, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "No GA billing id. Only shipped as gemini-2.0-pro-exp (experimental, free of charge); never had paid pricing. Not listed on official pricing page or LiteLLM catalog." - }, - "gemini-3-pro-preview": { - "input_per_mtok": 2.0, - "output_per_mtok": 12.0, - "cache_read_per_mtok": 0.2, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "Current 3.x flagship (page labels it Gemini 3.1 Pro Preview). Tiered: prompts >200k cost input $4.00, output $18.00, cache_read $0.40 per Mtok. Cache storage $4.50/Mtok/hr." - }, - "gemini-3-flash-preview": { - "input_per_mtok": 0.5, - "output_per_mtok": 3.0, - "cache_read_per_mtok": 0.05, - "cache_write_per_mtok": null, - "source": "https://ai.google.dev/gemini-api/docs/pricing", - "as_of": "2026-06", - "_note": "Text/image/video input; audio input $1.00, audio cache_read $0.10. Cache storage $1.00/Mtok/hr. No >200k context tier on standard." - } -} diff --git a/src/pricing/sources/openai.json b/src/pricing/sources/openai.json deleted file mode 100644 index 7b6b7d51..00000000 --- a/src/pricing/sources/openai.json +++ /dev/null @@ -1,106 +0,0 @@ -{ - "gpt-5": { - "input_per_mtok": 1.25, - "output_per_mtok": 10, - "cache_read_per_mtok": 0.125, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-5-mini": { - "input_per_mtok": 0.25, - "output_per_mtok": 2, - "cache_read_per_mtok": 0.025, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-5-nano": { - "input_per_mtok": 0.05, - "output_per_mtok": 0.4, - "cache_read_per_mtok": 0.005, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-5-codex": { - "input_per_mtok": 1.25, - "output_per_mtok": 10, - "cache_read_per_mtok": 0.125, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-4.1": { - "input_per_mtok": 2, - "output_per_mtok": 8, - "cache_read_per_mtok": 0.5, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-4.1-mini": { - "input_per_mtok": 0.4, - "output_per_mtok": 1.6, - "cache_read_per_mtok": 0.1, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-4.1-nano": { - "input_per_mtok": 0.1, - "output_per_mtok": 0.4, - "cache_read_per_mtok": 0.025, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-4o": { - "input_per_mtok": 2.5, - "output_per_mtok": 10, - "cache_read_per_mtok": 1.25, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "gpt-4o-mini": { - "input_per_mtok": 0.15, - "output_per_mtok": 0.6, - "cache_read_per_mtok": 0.075, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "o3": { - "input_per_mtok": 2, - "output_per_mtok": 8, - "cache_read_per_mtok": 0.5, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "o4-mini": { - "input_per_mtok": 1.1, - "output_per_mtok": 4.4, - "cache_read_per_mtok": 0.275, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "o3-mini": { - "input_per_mtok": 1.1, - "output_per_mtok": 4.4, - "cache_read_per_mtok": 0.55, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - }, - "codex-mini-latest": { - "input_per_mtok": 1.5, - "output_per_mtok": 6, - "cache_read_per_mtok": 0.375, - "cache_write_per_mtok": null, - "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json", - "as_of": "2026-06" - } -} diff --git a/src/pricing/sources/others.json b/src/pricing/sources/others.json deleted file mode 100644 index 1cdeb5e6..00000000 --- a/src/pricing/sources/others.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "grok-4": { - "input_per_mtok": 3, - "output_per_mtok": 15, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://docs.x.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Legacy id; xAI public pricing page now lists newer Grok versions (4.3/4.20/build-0.1). Rate verified via LiteLLM xai/ catalog (BerriAI/litellm) which sources directly from xAI." - }, - "grok-3": { - "input_per_mtok": 3, - "output_per_mtok": 15, - "cache_read_per_mtok": 0.75, - "cache_write_per_mtok": null, - "source": "https://docs.x.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Legacy id; rate from LiteLLM xai/grok-3 catalog (sourced from xAI)." - }, - "grok-code-fast-1": { - "input_per_mtok": 0.2, - "output_per_mtok": 1.5, - "cache_read_per_mtok": 0.02, - "cache_write_per_mtok": null, - "source": "https://docs.x.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Legacy id; rate from LiteLLM xai/grok-code-fast-1 catalog (sourced from xAI)." - }, - "grok-2": { - "input_per_mtok": 2, - "output_per_mtok": 10, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://docs.x.ai/docs/pricing", - "as_of": "2026-06", - "_note": "Legacy id (grok-2-1212); rate from LiteLLM xai/grok-2 catalog (sourced from xAI)." - }, - "mistral-large-latest": { - "input_per_mtok": 0.5, - "output_per_mtok": 1.5, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://mistral.ai/pricing", - "as_of": "2026-06", - "_note": "mistral-large-latest = Mistral Large 3." - }, - "codestral-latest": { - "input_per_mtok": 0.3, - "output_per_mtok": 0.9, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://mistral.ai/pricing", - "as_of": "2026-06", - "_note": "Codestral 2; price cross-confirmed via Google Vertex partner pricing ($0.30/$0.90)." - }, - "devstral": { - "input_per_mtok": 0.4, - "output_per_mtok": 2, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://mistral.ai/pricing", - "as_of": "2026-06", - "_note": "Maps to devstral-medium-latest (Mistral hosted API). devstral-small is $0.10/$0.30." - }, - "mistral-medium": { - "input_per_mtok": 0.4, - "output_per_mtok": 2, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://mistral.ai/pricing", - "as_of": "2026-06", - "_note": "Mistral Medium 3; cross-confirmed via Google Vertex partner pricing ($0.40/$2.00)." - }, - "llama-4-maverick": { - "input_per_mtok": 0.27, - "output_per_mtok": 0.85, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.together.ai/pricing", - "as_of": "2026-06", - "_note": "Host-dependent. Priced via Together AI serverless. Other hosts differ (e.g. AWS Bedrock, Vertex $0.35/$1.15)." - }, - "llama-4-scout": { - "input_per_mtok": 0.08, - "output_per_mtok": 0.3, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.together.ai/pricing", - "as_of": "2026-06", - "_note": "Host-dependent. Priced via Together AI serverless. Vertex lists $0.25/$0.70." - }, - "llama-3.3-70b": { - "input_per_mtok": 0.88, - "output_per_mtok": 0.88, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://www.together.ai/pricing", - "as_of": "2026-06", - "_note": "Host-dependent. Together Llama-3.3-70B-Instruct-Turbo ($0.88/$0.88, from LiteLLM together_ai source). Cheaper on DeepInfra ($0.13/$0.39); Vertex $0.72/$0.72." - }, - "command-a": { - "input_per_mtok": 2.5, - "output_per_mtok": 10, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://cohere.com/pricing", - "as_of": "2026-06", - "_note": "command-a-03-2025." - }, - "command-r-plus": { - "input_per_mtok": 2.5, - "output_per_mtok": 10, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://cohere.com/pricing", - "as_of": "2026-06", - "_note": "command-r-plus-08-2024 (legacy)." - }, - "amazon-nova-pro": { - "input_per_mtok": 0.8, - "output_per_mtok": 3.2, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://aws.amazon.com/bedrock/pricing/", - "as_of": "2026-06", - "_note": "AWS Bedrock on-demand ($0.0008/$0.0032 per 1K tokens)." - }, - "amazon-nova-lite": { - "input_per_mtok": 0.06, - "output_per_mtok": 0.24, - "cache_read_per_mtok": null, - "cache_write_per_mtok": null, - "source": "https://aws.amazon.com/bedrock/pricing/", - "as_of": "2026-06", - "_note": "AWS Bedrock on-demand ($0.00006/$0.00024 per 1K tokens)." - } -} diff --git a/src/session/extract.ts b/src/session/extract.ts index d6b4e857..1b0b0180 100644 --- a/src/session/extract.ts +++ b/src/session/extract.ts @@ -8,7 +8,7 @@ import { lookupPrice as catalogLookupPrice, computeCostUsd as catalogComputeCostUsd, -} from "../pricing/catalog.js"; +} from "./pricing.js"; // ── Public interfaces ────────────────────────────────────────────────────── @@ -30,6 +30,19 @@ export interface SessionEvent { * `Fetched and indexed N sections (XKB)` preamble. */ bytes_avoided?: number; + /** + * Optional structured cost/usage fields (Wave 2b). Emitted by + * extractAgentUsage alongside the colon-string `data` so the forward + * envelope can spread them to the platform as typed columns instead of an + * opaque blob. Present only when the source signal is present; cost_usd is + * omitted on a price miss or a zero-token turn. + */ + model_id?: string; + input_tokens?: number; + output_tokens?: number; + cache_read_tokens?: number; + cache_creation_tokens?: number; + cost_usd?: number; } export interface ToolCall { @@ -1487,21 +1500,39 @@ function extractAgentUsage(input: HookInput): SessionEvent[] { const cacheRead = typeof usage.cache_read_input_tokens === "number" ? usage.cache_read_input_tokens : 0; + const modelId = resolveModelId(input, out); const anyTokens = inputTokens > 0 || outputTokens > 0 || cacheCreate > 0 || cacheRead > 0; + let cost: number | null = null; if (anyTokens) { - const modelId = resolveModelId(input, out); // null ⇒ unmatched model id (catalog warned once) — skip the cost token // rather than blend a wrong Claude rate (the old non-Claude bug). - const cost = computeTurnCostUsd(modelId, inputTokens, outputTokens, cacheCreate, cacheRead); + cost = computeTurnCostUsd(modelId, inputTokens, outputTokens, cacheCreate, cacheRead); if (cost !== null) parts.push(`cost_usd:${formatCostUsd(cost)}`); } - return [{ + // Wave 2b — emit structured top-level fields alongside the colon-string so + // the forward envelope (which spreads `...event`) hands the platform typed + // columns. Each field is set only when its source signal is present, so the + // forward payload stays minimal; cost_usd is omitted on a price miss or a + // zero-token turn. The colon-string `data` stays for human/debug + back-compat. + const event: SessionEvent = { type: "agent_usage", category: "cost", data: safeString(parts.join(" ")), priority: 2, - }]; + }; + if (modelId.length > 0) event.model_id = modelId; + if (typeof usage.input_tokens === "number") event.input_tokens = usage.input_tokens; + if (typeof usage.output_tokens === "number") event.output_tokens = usage.output_tokens; + if (typeof usage.cache_read_input_tokens === "number") { + event.cache_read_tokens = usage.cache_read_input_tokens; + } + if (typeof usage.cache_creation_input_tokens === "number") { + event.cache_creation_tokens = usage.cache_creation_input_tokens; + } + if (cost !== null) event.cost_usd = cost; + + return [event]; } // ── User-message extractors ──────────────────────────────────────────────── diff --git a/src/session/model-prices.json b/src/session/model-prices.json new file mode 100644 index 00000000..1d70fc0f --- /dev/null +++ b/src/session/model-prices.json @@ -0,0 +1,429 @@ +{ + "claude-opus-4-8": { + "input_per_mtok": 5, + "output_per_mtok": 25, + "cache_read_per_mtok": 0.5, + "cache_write_per_mtok": 6.25, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-opus-4-7": { + "input_per_mtok": 5, + "output_per_mtok": 25, + "cache_read_per_mtok": 0.5, + "cache_write_per_mtok": 6.25, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-opus-4-6": { + "input_per_mtok": 5, + "output_per_mtok": 25, + "cache_read_per_mtok": 0.5, + "cache_write_per_mtok": 6.25, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-opus-4-5": { + "input_per_mtok": 5, + "output_per_mtok": 25, + "cache_read_per_mtok": 0.5, + "cache_write_per_mtok": 6.25, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-sonnet-4-6": { + "input_per_mtok": 3, + "output_per_mtok": 15, + "cache_read_per_mtok": 0.3, + "cache_write_per_mtok": 3.75, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-sonnet-4-5": { + "input_per_mtok": 3, + "output_per_mtok": 15, + "cache_read_per_mtok": 0.3, + "cache_write_per_mtok": 3.75, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-haiku-4-5": { + "input_per_mtok": 1, + "output_per_mtok": 5, + "cache_read_per_mtok": 0.1, + "cache_write_per_mtok": 1.25, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-3-7-sonnet": { + "input_per_mtok": 3, + "output_per_mtok": 15, + "cache_read_per_mtok": 0.3, + "cache_write_per_mtok": 3.75, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "claude-3-5-haiku": { + "input_per_mtok": 0.8, + "output_per_mtok": 4, + "cache_read_per_mtok": 0.08, + "cache_write_per_mtok": 1, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "claude-fable-5": { + "input_per_mtok": 10, + "output_per_mtok": 50, + "cache_read_per_mtok": 1, + "cache_write_per_mtok": 12.5, + "source": "https://platform.claude.com/docs/en/about-claude/pricing" + }, + "gpt-5": { + "input_per_mtok": 1.25, + "output_per_mtok": 10, + "cache_read_per_mtok": 0.125, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-5-mini": { + "input_per_mtok": 0.25, + "output_per_mtok": 2, + "cache_read_per_mtok": 0.025, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-5-nano": { + "input_per_mtok": 0.05, + "output_per_mtok": 0.4, + "cache_read_per_mtok": 0.005, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-5-codex": { + "input_per_mtok": 1.25, + "output_per_mtok": 10, + "cache_read_per_mtok": 0.125, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-4.1": { + "input_per_mtok": 2, + "output_per_mtok": 8, + "cache_read_per_mtok": 0.5, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-4.1-mini": { + "input_per_mtok": 0.4, + "output_per_mtok": 1.6, + "cache_read_per_mtok": 0.1, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-4.1-nano": { + "input_per_mtok": 0.1, + "output_per_mtok": 0.4, + "cache_read_per_mtok": 0.025, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-4o": { + "input_per_mtok": 2.5, + "output_per_mtok": 10, + "cache_read_per_mtok": 1.25, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gpt-4o-mini": { + "input_per_mtok": 0.15, + "output_per_mtok": 0.6, + "cache_read_per_mtok": 0.075, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "o3": { + "input_per_mtok": 2, + "output_per_mtok": 8, + "cache_read_per_mtok": 0.5, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "o4-mini": { + "input_per_mtok": 1.1, + "output_per_mtok": 4.4, + "cache_read_per_mtok": 0.275, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "o3-mini": { + "input_per_mtok": 1.1, + "output_per_mtok": 4.4, + "cache_read_per_mtok": 0.55, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "codex-mini-latest": { + "input_per_mtok": 1.5, + "output_per_mtok": 6, + "cache_read_per_mtok": 0.375, + "cache_write_per_mtok": null, + "source": "https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json" + }, + "gemini-2.5-pro": { + "input_per_mtok": 1.25, + "output_per_mtok": 10, + "cache_read_per_mtok": 0.125, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "gemini-2.5-flash": { + "input_per_mtok": 0.3, + "output_per_mtok": 2.5, + "cache_read_per_mtok": 0.03, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "gemini-2.5-flash-lite": { + "input_per_mtok": 0.1, + "output_per_mtok": 0.4, + "cache_read_per_mtok": 0.01, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "gemini-2.0-flash": { + "input_per_mtok": 0.1, + "output_per_mtok": 0.4, + "cache_read_per_mtok": 0.025, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "gemini-2.0-flash-lite": { + "input_per_mtok": 0.075, + "output_per_mtok": 0.3, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "gemini-3-pro-preview": { + "input_per_mtok": 2, + "output_per_mtok": 12, + "cache_read_per_mtok": 0.2, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "gemini-3-flash-preview": { + "input_per_mtok": 0.5, + "output_per_mtok": 3, + "cache_read_per_mtok": 0.05, + "cache_write_per_mtok": null, + "source": "https://ai.google.dev/gemini-api/docs/pricing" + }, + "qwen3-coder": { + "input_per_mtok": 1, + "output_per_mtok": 5, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.alibabacloud.com/help/en/model-studio/models" + }, + "qwen-max": { + "input_per_mtok": 1.6, + "output_per_mtok": 6.4, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.alibabacloud.com/help/en/model-studio/models" + }, + "qwen-plus": { + "input_per_mtok": 0.4, + "output_per_mtok": 1.2, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.alibabacloud.com/help/en/model-studio/models" + }, + "qwen-turbo": { + "input_per_mtok": 0.05, + "output_per_mtok": 0.2, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.alibabacloud.com/help/en/model-studio/models" + }, + "qwen3-max": { + "input_per_mtok": 1.2, + "output_per_mtok": 6, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.alibabacloud.com/help/en/model-studio/models" + }, + "kimi-k2": { + "input_per_mtok": 0.6, + "output_per_mtok": 2.5, + "cache_read_per_mtok": 0.15, + "cache_write_per_mtok": null, + "source": "https://platform.moonshot.ai/docs/pricing/chat" + }, + "kimi-k2-turbo": { + "input_per_mtok": 1.15, + "output_per_mtok": 8, + "cache_read_per_mtok": 0.15, + "cache_write_per_mtok": null, + "source": "https://platform.moonshot.ai/docs/pricing/chat" + }, + "moonshot-v1-8k": { + "input_per_mtok": 0.2, + "output_per_mtok": 2, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://platform.moonshot.ai/docs/pricing" + }, + "moonshot-v1-32k": { + "input_per_mtok": 1, + "output_per_mtok": 3, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://platform.moonshot.ai/docs/pricing" + }, + "moonshot-v1-128k": { + "input_per_mtok": 2, + "output_per_mtok": 5, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://platform.moonshot.ai/docs/pricing" + }, + "deepseek-v3": { + "input_per_mtok": 0.27, + "output_per_mtok": 1.1, + "cache_read_per_mtok": 0.07, + "cache_write_per_mtok": 0, + "source": "https://api-docs.deepseek.com/quick_start/pricing" + }, + "deepseek-r1": { + "input_per_mtok": 0.55, + "output_per_mtok": 2.19, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://api-docs.deepseek.com/quick_start/pricing" + }, + "deepseek-chat": { + "input_per_mtok": 0.14, + "output_per_mtok": 0.28, + "cache_read_per_mtok": 0.0028, + "cache_write_per_mtok": null, + "source": "https://api-docs.deepseek.com/quick_start/pricing" + }, + "deepseek-reasoner": { + "input_per_mtok": 0.14, + "output_per_mtok": 0.28, + "cache_read_per_mtok": 0.0028, + "cache_write_per_mtok": null, + "source": "https://api-docs.deepseek.com/quick_start/pricing" + }, + "glm-4.6": { + "input_per_mtok": 0.6, + "output_per_mtok": 2.2, + "cache_read_per_mtok": 0.11, + "cache_write_per_mtok": null, + "source": "https://docs.z.ai/guides/overview/pricing" + }, + "glm-4-air": { + "input_per_mtok": 0.2, + "output_per_mtok": 1.1, + "cache_read_per_mtok": 0.03, + "cache_write_per_mtok": null, + "source": "https://docs.z.ai/guides/overview/pricing" + }, + "grok-4": { + "input_per_mtok": 3, + "output_per_mtok": 15, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://docs.x.ai/docs/pricing" + }, + "grok-3": { + "input_per_mtok": 3, + "output_per_mtok": 15, + "cache_read_per_mtok": 0.75, + "cache_write_per_mtok": null, + "source": "https://docs.x.ai/docs/pricing" + }, + "grok-code-fast-1": { + "input_per_mtok": 0.2, + "output_per_mtok": 1.5, + "cache_read_per_mtok": 0.02, + "cache_write_per_mtok": null, + "source": "https://docs.x.ai/docs/pricing" + }, + "grok-2": { + "input_per_mtok": 2, + "output_per_mtok": 10, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://docs.x.ai/docs/pricing" + }, + "mistral-large-latest": { + "input_per_mtok": 0.5, + "output_per_mtok": 1.5, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://mistral.ai/pricing" + }, + "codestral-latest": { + "input_per_mtok": 0.3, + "output_per_mtok": 0.9, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://mistral.ai/pricing" + }, + "devstral": { + "input_per_mtok": 0.4, + "output_per_mtok": 2, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://mistral.ai/pricing" + }, + "mistral-medium": { + "input_per_mtok": 0.4, + "output_per_mtok": 2, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://mistral.ai/pricing" + }, + "llama-4-maverick": { + "input_per_mtok": 0.27, + "output_per_mtok": 0.85, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.together.ai/pricing" + }, + "llama-4-scout": { + "input_per_mtok": 0.08, + "output_per_mtok": 0.3, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.together.ai/pricing" + }, + "llama-3.3-70b": { + "input_per_mtok": 0.88, + "output_per_mtok": 0.88, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://www.together.ai/pricing" + }, + "command-a": { + "input_per_mtok": 2.5, + "output_per_mtok": 10, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://cohere.com/pricing" + }, + "command-r-plus": { + "input_per_mtok": 2.5, + "output_per_mtok": 10, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://cohere.com/pricing" + }, + "amazon-nova-pro": { + "input_per_mtok": 0.8, + "output_per_mtok": 3.2, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://aws.amazon.com/bedrock/pricing/" + }, + "amazon-nova-lite": { + "input_per_mtok": 0.06, + "output_per_mtok": 0.24, + "cache_read_per_mtok": null, + "cache_write_per_mtok": null, + "source": "https://aws.amazon.com/bedrock/pricing/" + } +} diff --git a/src/pricing/catalog.ts b/src/session/pricing.ts similarity index 75% rename from src/pricing/catalog.ts rename to src/session/pricing.ts index fff0275a..cb9faa8e 100644 --- a/src/pricing/catalog.ts +++ b/src/session/pricing.ts @@ -1,9 +1,9 @@ /** * Pricing catalog — single source of truth for per-model USD cost. * - * Deep module, tiny interface. At load it merges the 5 curated vendor JSONs - * (src/pricing/sources/{anthropic,openai,google,chinese,others}.json) into one - * Map in per-Mtok units, then exposes three pure-ish functions: + * Deep module, tiny interface. At load it reads one curated multi-vendor JSON + * (src/session/model-prices.json) into a Map in per-Mtok units, + * then exposes three pure-ish functions: * * lookupPrice(modelId) → Price | null * computeCostUsd(modelId, tokens) → number | null @@ -17,18 +17,14 @@ * an unknown id resolves to `null` (no price) instead of a wrong Claude rate. * * The large litellm catalog (~1.5MB, ~2900 models) is NOT bundled — it lives at - * tools/pricing/litellm-catalog.json as the dev-only refresh base for these - * curated JSONs. See tools/pricing/litellm-NOTES.md for the refresh recipe. + * tools/pricing/litellm-catalog.json as the dev-only refresh base for this + * curated JSON. * - * The 5 curated JSONs are small (~25KB total) and esbuild inlines them into the + * The curated JSON is small (~13KB, 61 models) and esbuild inlines it into the * hook/server bundles at build time (no runtime fs read, no external file). */ -import anthropic from "./sources/anthropic.json" with { type: "json" }; -import openai from "./sources/openai.json" with { type: "json" }; -import google from "./sources/google.json" with { type: "json" }; -import chinese from "./sources/chinese.json" with { type: "json" }; -import others from "./sources/others.json" with { type: "json" }; +import catalog from "./model-prices.json" with { type: "json" }; /** Per-Mtok price for one model. Any of the four rates may be null ("unknown"). */ export interface Price { @@ -46,7 +42,7 @@ export interface TokenCounts { cache_creation_tokens?: number; } -/** Raw shape of a curated source row (carries provenance fields we drop). */ +/** Raw shape of a curated source row (carries a provenance `source` we drop). */ interface RawRow { input_per_mtok: number | null; output_per_mtok: number | null; @@ -56,36 +52,28 @@ interface RawRow { } /** - * Merge the 5 vendor JSONs into one Map. A row with a null *input* price is + * Read the curated JSON into one Map. A row with a null *input* price is * unusable for cost (the primary bucket has no rate) and is dropped at load so * lookupPrice returns null for it — matching "null-priced entries → no price". - * Curated ids are globally unique across the five files (verified), so merge - * order is irrelevant; later files would otherwise win on collision. + * (The two null-input ids are already pruned from the JSON itself; this guard + * keeps the loader robust if one is ever re-added.) */ function buildCatalog(): Map { const map = new Map(); - const sources: Record[] = [ - anthropic as Record, - openai as Record, - google as Record, - chinese as Record, - others as Record, - ]; - for (const src of sources) { - for (const id of Object.keys(src)) { - const row = src[id]; - if (row == null || typeof row !== "object") continue; - // No input rate ⇒ no usable price for this model. - if (typeof row.input_per_mtok !== "number") continue; - map.set(id, { - input_per_mtok: row.input_per_mtok, - output_per_mtok: typeof row.output_per_mtok === "number" ? row.output_per_mtok : null, - cache_read_per_mtok: - typeof row.cache_read_per_mtok === "number" ? row.cache_read_per_mtok : null, - cache_write_per_mtok: - typeof row.cache_write_per_mtok === "number" ? row.cache_write_per_mtok : null, - }); - } + const src = catalog as Record; + for (const id of Object.keys(src)) { + const row = src[id]; + if (row == null || typeof row !== "object") continue; + // No input rate ⇒ no usable price for this model. + if (typeof row.input_per_mtok !== "number") continue; + map.set(id, { + input_per_mtok: row.input_per_mtok, + output_per_mtok: typeof row.output_per_mtok === "number" ? row.output_per_mtok : null, + cache_read_per_mtok: + typeof row.cache_read_per_mtok === "number" ? row.cache_read_per_mtok : null, + cache_write_per_mtok: + typeof row.cache_write_per_mtok === "number" ? row.cache_write_per_mtok : null, + }); } return map; } diff --git a/tests/session/extract-agent-usage.test.ts b/tests/session/extract-agent-usage.test.ts index 5ae161da..85ac0b4d 100644 --- a/tests/session/extract-agent-usage.test.ts +++ b/tests/session/extract-agent-usage.test.ts @@ -292,3 +292,90 @@ describe("extractAgentUsage — Issue #4 AgentOutput.usage capture", () => { } }); }); + +/** + * Wave 2b — structured cost event. + * + * The colon-string `data` is opaque to the platform (it cannot column-ize a + * "tokens_in:123 cost_usd:0.02" blob). extractAgentUsage now also emits the + * cost/token signals as top-level SessionEvent fields, which the forward + * envelope spreads straight to the platform as typed columns: + * + * model_id, input_tokens, output_tokens, + * cache_read_tokens, cache_creation_tokens, cost_usd + * + * The colon-string `data` stays for human/debug + back-compat. + */ +describe("extractAgentUsage — Wave 2b structured cost fields", () => { + function usageEvent(toolInput: Record, usage: Record, extra: Record = {}) { + return extractEvents({ + tool_name: "Task", + tool_input: toolInput, + tool_response: JSON.stringify({ ...extra, usage }), + }).filter((e) => e.type === "agent_usage")[0]; + } + + // (a) the 6 structured fields ride the event with correct values + test("(a) Task usage yields event carrying the 6 structured fields with correct values", () => { + const ev = usageEvent( + { model: "claude-sonnet-4-6" }, + { + input_tokens: 1000, + output_tokens: 500, + cache_creation_input_tokens: 1000, + cache_read_input_tokens: 1500, + }, + ); + expect(ev.model_id).toBe("claude-sonnet-4-6"); + expect(ev.input_tokens).toBe(1000); + expect(ev.output_tokens).toBe(500); + expect(ev.cache_creation_tokens).toBe(1000); + expect(ev.cache_read_tokens).toBe(1500); + // 1000*3 + 500*15 + 1000*3.75 + 1500*0.30 = 14700 / 1e6 = 0.0147 + expect(ev.cost_usd).toBeCloseTo(0.0147, 8); + }); + + // (b) cost_usd matches the catalog for a known model + test("(b) cost_usd matches the catalog for a known model (gpt-5)", () => { + const ev = usageEvent( + { model: "gpt-5" }, + { input_tokens: 1000, output_tokens: 500 }, + ); + // gpt-5: 1000*1.25 + 500*10 = 6250 / 1e6 = 0.00625 + expect(ev.cost_usd).toBeCloseTo(0.00625, 8); + expect(ev.model_id).toBe("gpt-5"); + }); + + // (c) unknown model → tokens present, cost_usd omitted (no Claude fallback) + test("(c) unknown model → tokens present, cost_usd omitted/null", () => { + const warn = vi.spyOn(console, "warn").mockImplementation(() => {}); + const ev = usageEvent( + { model: "claude-future-model-99" }, + { input_tokens: 1000, output_tokens: 500 }, + ); + warn.mockRestore(); + expect(ev.input_tokens).toBe(1000); + expect(ev.output_tokens).toBe(500); + expect(ev.cost_usd == null).toBe(true); + }); + + // (d) zero-token response → no cost_usd + test("(d) zero-token response → no cost_usd", () => { + const ev = usageEvent( + { model: "claude-sonnet-4-6" }, + { input_tokens: 0, output_tokens: 0 }, + ); + expect(ev.cost_usd == null).toBe(true); + }); + + // (e) the existing colon-string `data` still present for back-compat + test("(e) colon-string data still present for back-compat", () => { + const ev = usageEvent( + { model: "claude-sonnet-4-6" }, + { input_tokens: 1000, output_tokens: 500 }, + ); + expect(ev.data).toMatch(/tokens_in:1000/); + expect(ev.data).toMatch(/tokens_out:500/); + expect(ev.data).toMatch(/cost_usd:0\.0105/); + }); +}); diff --git a/tests/pricing/catalog.test.ts b/tests/session/pricing.test.ts similarity index 95% rename from tests/pricing/catalog.test.ts rename to tests/session/pricing.test.ts index dc5d522f..84fd48e9 100644 --- a/tests/pricing/catalog.test.ts +++ b/tests/session/pricing.test.ts @@ -6,7 +6,7 @@ * a Claude rate (non-Claude models silently inherited Sonnet's default), * which over-/under-charged every OpenAI / Gemini / Qwen / DeepSeek turn. * - * The catalog merges the 5 curated vendor JSONs (src/pricing/sources/*.json) + * The catalog reads the curated multi-vendor JSON (src/session/model-prices.json) * into one per-Mtok price map and prices each model from ITS OWN row. * Behaviours under test: * (a) curated lookup hits across all five vendors @@ -23,9 +23,9 @@ import { lookupPrice, computeCostUsd, nativeOrComputed, -} from "../../src/pricing/catalog.js"; +} from "../../src/session/pricing.js"; -describe("pricing/catalog — lookupPrice", () => { +describe("session/pricing — lookupPrice", () => { // (a) curated lookup hits test("(a) Anthropic curated hit returns per-Mtok price", () => { const p = lookupPrice("claude-opus-4-8"); @@ -79,7 +79,7 @@ describe("pricing/catalog — lookupPrice", () => { }); }); -describe("pricing/catalog — computeCostUsd", () => { +describe("session/pricing — computeCostUsd", () => { // (c) THE BUG: non-Claude model must use its own price, not Claude's. test("(c) gpt-5 priced from its own row, NOT Claude default", () => { const tokens = { input_tokens: 1000, output_tokens: 500 }; @@ -157,7 +157,7 @@ describe("pricing/catalog — computeCostUsd", () => { }); }); -describe("pricing/catalog — nativeOrComputed", () => { +describe("session/pricing — nativeOrComputed", () => { // (g) native-cost passthrough test("(g) provider native cost wins over computed", () => { const native = nativeOrComputed("gpt-5", { input_tokens: 1000 }, 0.42); diff --git a/tools/pricing/litellm-NOTES.md b/tools/pricing/litellm-NOTES.md deleted file mode 100644 index 954e216f..00000000 --- a/tools/pricing/litellm-NOTES.md +++ /dev/null @@ -1,80 +0,0 @@ -# LiteLLM Catalog — Adapter Notes - -Vendored from BerriAI/litellm as context-mode's comprehensive pricing base, so -unknown/unseen models still resolve a price instead of failing. - -- **Source:** `https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json` -- **Local file:** `litellm-catalog.json` -- **Size:** ~1.53 MB (1,570,159 bytes) -- **Top-level keys:** 2,901 (2,900 model entries + 1 `sample_spec` template — skip `sample_spec`) -- **Entries with numeric `input_cost_per_token`:** 2,432 -- **Distinct `litellm_provider` values:** 120 (top: fireworks_ai, bedrock, openai, azure, gemini, mistral, openrouter …) - -## Shape - -The file is a flat JSON object. **Each key is a model id**; each value is a metadata object. -There is no wrapper array. One special key, `sample_spec`, is a documentation template -(not a real model) and must be excluded. - -```jsonc -{ - "sample_spec": { /* template — ignore */ }, - "gpt-4o": { "input_cost_per_token": 0.0000025, ... }, - "claude-sonnet-4-20250514": { ... } -} -``` - -## Cost fields — verified present in the fetched JSON - -> All `*_cost_per_token` values are **PER TOKEN** (USD), not per-million. -> To convert to per-Mtok (context-mode's internal unit): **multiply by `1e6` (×1,000,000).** -> e.g. `gpt-4o` `input_cost_per_token = 0.0000025` → `0.0000025 × 1e6 = $2.50 / Mtok`. - -Core fields (use these; coverage count in parentheses): - -| Field | Meaning | Count | -|-------|---------|-------| -| `input_cost_per_token` | prompt cost per token | 2,433 | -| `output_cost_per_token` | completion cost per token | 2,430 | -| `cache_read_input_token_cost` | cached-prompt read cost per token | 645 | -| `cache_creation_input_token_cost` | cache-write/creation cost per token | 210 | - -Real example (`claude-sonnet-4-20250514`, provider `anthropic`): -`input_cost_per_token: 0.000003` (= $3.00/Mtok), `output_cost_per_token: 0.000015` (= $15.00/Mtok), -`cache_read_input_token_cost: 3e-7` (= $0.30/Mtok), `cache_creation_input_token_cost: 0.00000375` (= $3.75/Mtok). - -### Schema surprises / gotchas - -- **Tiered & variant cost fields exist** — e.g. `input_cost_per_token_above_200k_tokens`, - `cache_read_input_token_cost_above_200k_tokens`, `_priority`, `_batches`, `_flex`, - `_above_1hr`, `_above_272k_tokens`. Treat these as optional overrides; fall back to the - base `input_cost_per_token` / `output_cost_per_token`. -- **Non-token cost units also appear** and are NOT per-token: `input_cost_per_second`, - `input_cost_per_character`, `input_cost_per_image`, `input_cost_per_pixel`, - `output_cost_per_second`, `output_cost_per_reasoning_token`, - `search_context_cost_per_query` (the last can be an object keyed by search-context size). - Do not blindly ×1e6 these — only the `*_per_token` family is per-token. -- **2,366 of 2,900 keys contain `/`** — namespaced ids like `bedrock/...`, - `1024-x-1024/dall-e-2`, image/resolution-prefixed entries. Match on the full key. -- Metadata fields used for context limits: `max_tokens`, `max_input_tokens`, - `max_output_tokens` (and `mode` distinguishes `chat`, `embedding`, image, etc.). -- Some entries carry `deprecation_date` (81 entries). - -## Lookup strategy (for the adapter) - -Given an incoming `model_id`: - -1. **Exact match** — `catalog[model_id]`. Fastest; covers the common case. -2. **Provider-stripped / namespaced fallback** — many ids are `provider/model`. - Try stripping or adding a known provider prefix: - - if `model_id` has no `/`, try `catalog[provider + "/" + model_id]`; - - if `model_id` is `provider/model`, also try the bare `model` segment. -3. **Provider + model match** — scan entries whose `litellm_provider` matches the resolved - provider and whose key endsWith the model segment. -4. **Skip `sample_spec`** in every scan. -5. From the matched entry, read `input_cost_per_token` / `output_cost_per_token` - (+ optional `cache_read_input_token_cost`, `cache_creation_input_token_cost`), - then **× 1e6** to get per-Mtok rates. -6. If nothing matches or `input_cost_per_token` is absent (some entries price only by - second/character/image), the model has no usable per-token price — fall through to - context-mode's own default.