feat(ai): add GPT-6 Astra support

This commit is contained in:
Armin Ronacher
2026-09-04 22:08:06 +02:00
parent 92d8e2d17d
commit 17de82d7be
8 changed files with 91 additions and 22 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added GPT-6 Astra for OpenAI API keys and OpenAI Codex subscriptions.
## [0.85.0] - 2026-09-04
### Breaking Changes
+53 -4
View File
@@ -354,13 +354,19 @@ const OPENAI_TOOL_SEARCH_MODEL_IDS = new Set([
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-6-astra",
]);
// Public OpenAI documents additional_tools for applications that load tools
// outside the normal tool-search flow. Codex currently uses the input item for
// its Responses Lite GPT-5.6 models.
// https://developers.openai.com/api/docs/guides/tools-tool-search#add-tools-at-a-specific-point-in-the-input
const OPENAI_ADDITIONAL_TOOLS_MODEL_IDS = OPENAI_TOOL_SEARCH_MODEL_IDS;
const OPENAI_CODEX_ADDITIONAL_TOOLS_MODEL_IDS = new Set(["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"]);
const OPENAI_CODEX_ADDITIONAL_TOOLS_MODEL_IDS = new Set([
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-6-astra",
]);
const OPENAI_LONG_CONTEXT_INPUT_THRESHOLD = 272000;
const OPENAI_SHORT_CONTEXT_CAPPED_MODEL_IDS = new Set([
"gpt-5.4",
@@ -368,6 +374,7 @@ const OPENAI_SHORT_CONTEXT_CAPPED_MODEL_IDS = new Set([
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-6-astra",
]);
const OPENAI_LONG_CONTEXT_PRICING_MODEL_IDS = new Set([
"gpt-5.4",
@@ -377,6 +384,7 @@ const OPENAI_LONG_CONTEXT_PRICING_MODEL_IDS = new Set([
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-6-astra",
]);
function withOpenAiLongContextPricing(cost: Model<Api>["cost"]): Model<Api>["cost"] {
@@ -525,13 +533,14 @@ function supportsOpenAiXhigh(modelId: string): boolean {
modelId.includes("gpt-5.3") ||
modelId.includes("gpt-5.4") ||
modelId.includes("gpt-5.5") ||
modelId.includes("gpt-5.6")
modelId.includes("gpt-5.6") ||
modelId.includes("gpt-6-astra")
);
}
function supportsOpenAiMax(model: Model<Api>): boolean {
return (
model.id.includes("gpt-5.6") &&
(model.id.includes("gpt-5.6") || model.id.includes("gpt-6-astra")) &&
(model.api === "openai-responses" ||
model.api === "azure-openai-responses" ||
model.api === "openai-codex-responses" ||
@@ -867,6 +876,22 @@ function applyThinkingLevelMetadata(model: Model<any>): void {
) {
mergeThinkingLevelMap(model, { off: null });
}
if (
model.id === "gpt-6-astra" &&
(model.api === "openai-responses" ||
model.api === "azure-openai-responses" ||
model.api === "openai-codex-responses")
) {
mergeThinkingLevelMap(model, {
off: null,
minimal: null,
low: "low",
medium: "medium",
high: "high",
xhigh: "xhigh",
max: "max",
});
}
if (model.provider === "github-copilot" && model.id.startsWith("gpt-5")) {
mergeThinkingLevelMap(model, { minimal: "low" });
}
@@ -2488,6 +2513,18 @@ async function generateModels() {
// Add missing gpt models
const missingOpenAiModels: Model<"openai-responses">[] = [
{
id: "gpt-6-astra",
name: "GPT-6 Astra",
api: "openai-responses",
baseUrl: "https://api.openai.com/v1",
provider: "openai",
reasoning: true,
input: ["text", "image"],
cost: withOpenAiLongContextPricing({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }),
contextWindow: OPENAI_LONG_CONTEXT_INPUT_THRESHOLD,
maxTokens: 128000,
},
{
id: "gpt-5.6-sol",
name: "GPT-5.6 Sol",
@@ -2693,13 +2730,25 @@ async function generateModels() {
// OpenAI Codex (ChatGPT OAuth) models
// NOTE: These are not fetched from models.dev; we keep a small, explicit list to avoid aliases.
// Older model limits are based on observed server behavior; GPT-5.6 follows Codex's 272k catalog limit (formerly 372k).
// Older model limits are based on observed server behavior; GPT-5.6 and GPT-6 Astra use Codex's 272k default catalog limit.
const CODEX_BASE_URL = "https://chatgpt.com/backend-api";
const CODEX_CONTEXT = 272000;
const CODEX_GPT_56_CONTEXT = 272000;
const CODEX_SPARK_CONTEXT = 128000;
const CODEX_MAX_TOKENS = 128000;
const codexModels: Model<"openai-codex-responses">[] = [
{
id: "gpt-6-astra",
name: "GPT-6 Astra",
api: "openai-codex-responses",
provider: "openai-codex",
baseUrl: CODEX_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: withOpenAiLongContextPricing({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }),
contextWindow: CODEX_CONTEXT,
maxTokens: CODEX_MAX_TOKENS,
},
{
id: "gpt-5.3-codex-spark",
name: "GPT-5.3 Codex Spark",
+17 -4
View File
@@ -83,7 +83,19 @@ function getPromptCacheRetention(
compat: Required<OpenAIResponsesCompat>,
cacheRetention: CacheRetention,
): "24h" | undefined {
return cacheRetention === "long" && compat.supportsLongCacheRetention ? "24h" : undefined;
return cacheRetention === "long" && compat.supportsLongCacheRetention && !compat.supportsExplicitPromptCacheMode
? "24h"
: undefined;
}
function getPromptCacheOptions(
compat: Required<OpenAIResponsesCompat>,
cacheRetention: CacheRetention,
): { mode?: "explicit"; ttl?: "30m" } | undefined {
if (!compat.supportsExplicitPromptCacheMode) return undefined;
if (cacheRetention === "none") return { mode: "explicit" };
if (cacheRetention === "long" && compat.supportsLongCacheRetention) return { ttl: "30m" };
return undefined;
}
function formatOpenAIResponsesError(error: unknown): string {
@@ -287,14 +299,15 @@ function buildParams(
});
const cacheRetention = resolveCacheRetention(options?.cacheRetention, options?.env);
const disableImplicitPromptCache = cacheRetention === "none" && compat.supportsExplicitPromptCacheMode;
const params: ResponseCreateParamsStreaming & { prompt_cache_options?: { mode: "explicit" } } = {
const params: ResponseCreateParamsStreaming & {
prompt_cache_options?: { mode?: "explicit"; ttl?: "30m" };
} = {
model: model.id,
input: messages,
stream: true,
prompt_cache_key: cacheRetention === "none" ? undefined : clampOpenAIPromptCacheKey(options?.sessionId),
prompt_cache_retention: getPromptCacheRetention(compat, cacheRetention),
prompt_cache_options: disableImplicitPromptCache ? { mode: "explicit" } : undefined,
prompt_cache_options: getPromptCacheOptions(compat, cacheRetention),
store: false,
};
+2 -2
View File
@@ -648,7 +648,7 @@ export interface OpenAIResponsesCompat {
supportsDeveloperRole?: boolean;
/** Session-affinity header format: `openai` sends `session_id` and `x-client-request-id`; `openai-nosession` sends `x-client-request-id`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */
sessionAffinityFormat?: SessionAffinityFormat;
/** Whether the provider supports `prompt_cache_retention: "24h"`. Default: true. */
/** Whether the provider supports long prompt cache retention. This uses `prompt_cache_options.ttl: "30m"` on GPT-5.6+ and `prompt_cache_retention: "24h"` on earlier models. Default: true. */
supportsLongCacheRetention?: boolean;
/** Whether the provider supports strict JSON-schema function tools. Defaults are API-specific; generated OpenAI models enable it explicitly. */
supportsStrictMode?: boolean;
@@ -658,7 +658,7 @@ export interface OpenAIResponsesCompat {
supportsAdditionalTools?: boolean;
/** Whether the model supports client-executed tool search for deferred tools. Default: false. */
supportsToolSearch?: boolean;
/** Whether the model accepts `prompt_cache_options` (OpenAI GPT-5.6+ explicit prompt caching). Older OpenAI models reject the parameter. Default: false. */
/** Whether the model accepts `prompt_cache_options` (OpenAI GPT-5.6+ prompt caching). Older OpenAI models reject the parameter. Default: false. */
supportsExplicitPromptCacheMode?: boolean;
/** Whether the provider accepts the `max_output_tokens` parameter. Some Codex-protocol gateways reject it. Default: true. */
supportsMaxOutputTokens?: boolean;
+11 -8
View File
@@ -19,7 +19,7 @@ interface OpenAICompletionsCachePayload {
}
interface OpenAIResponsesCachePayload extends OpenAICompletionsCachePayload {
prompt_cache_options?: { mode: "explicit" };
prompt_cache_options?: { mode?: "explicit"; ttl?: "30m" };
}
function stopAfterPayload<TPayload>(capture: (payload: TPayload) => void): (payload: unknown) => never {
@@ -398,16 +398,19 @@ describe("Cache Retention (PI_CACHE_RETENTION)", () => {
expect(capturedPayload?.prompt_cache_options).toBeUndefined();
});
it("should set prompt_cache_retention when cacheRetention is long", async () => {
const model = getModel("openai", "gpt-4o-mini");
let capturedPayload: any = null;
it.each([
["gpt-4o-mini", "24h", undefined],
["gpt-6-astra", undefined, { ttl: "30m" }],
] as const)("should use the supported long cache field for %s", async (modelId, retention, cacheOptions) => {
const model = getModel("openai", modelId);
let capturedPayload: OpenAIResponsesCachePayload | undefined;
try {
const s = streamOpenAIResponses(model, context, {
apiKey: "fake-key",
cacheRetention: "long",
sessionId: "session-2",
onPayload: stopAfterPayload((payload) => {
onPayload: stopAfterPayload<OpenAIResponsesCachePayload>((payload) => {
capturedPayload = payload;
}),
});
@@ -419,9 +422,9 @@ describe("Cache Retention (PI_CACHE_RETENTION)", () => {
// Expected to fail
}
expect(capturedPayload).not.toBeNull();
expect(capturedPayload.prompt_cache_key).toBe("session-2");
expect(capturedPayload.prompt_cache_retention).toBe("24h");
expect(capturedPayload?.prompt_cache_key).toBe("session-2");
expect(capturedPayload?.prompt_cache_retention).toBe(retention);
expect(capturedPayload?.prompt_cache_options).toEqual(cacheOptions);
});
});
+2 -2
View File
@@ -67,8 +67,8 @@ describe("max thinking level", () => {
expect(clampThinkingLevel(model, "xhigh")).toBe("max");
});
it("sends max to the Codex Responses API", async () => {
const model = getModel("openai-codex", "gpt-5.6-sol")!;
it.each(["gpt-5.6-sol", "gpt-6-astra"] as const)("sends max to the Codex Responses API for %s", async (modelId) => {
const model = getModel("openai-codex", modelId)!;
const context: Context = {
systemPrompt: "You are a helpful assistant.",
messages: [{ role: "user", content: "Hello", timestamp: Date.now() }],
+1 -1
View File
@@ -52,7 +52,7 @@ describe("getSupportedThinkingLevels", () => {
expect(getSupportedThinkingLevels(model!)).not.toContain("max");
});
it.each(["gpt-5.4", "gpt-5.5", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"] as const)(
it.each(["gpt-5.4", "gpt-5.5", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", "gpt-6-astra"] as const)(
"includes xhigh for openai-codex %s models",
(modelId) => {
const model = getModel("openai-codex", modelId);
+1 -1
View File
@@ -482,7 +482,7 @@ For providers with partial OpenAI compatibility, use the `compat` field.
| `supportsStrictMode` | Whether the provider accepts strict JSON-schema function tool definitions. Defaults depend on the API; built-in OpenAI models carry explicit capability metadata. |
| `supportsOpenAIGrammarTools` | Whether OpenAI-compatible APIs emit custom Lark/regex grammar tools. When `false`, grammar-constrained tools fall back to normal function tools. Default: `false`; the built-in model catalog enables it for GPT-5+ models on OpenAI, OpenAI Codex, Azure OpenAI, GitHub Copilot, opencode, and Cloudflare AI Gateway. |
| `deferredToolsMode` | Use provider-specific deferred tool serialization. Currently only `"kimi"` is supported for Kimi's OpenAI-compatible Chat Completions format. |
| `supportsLongCacheRetention` | Whether the provider accepts long cache retention when cache retention is `long`: `prompt_cache_retention: "24h"` for OpenAI prompt caching, or `cache_control.ttl: "1h"` when `cacheControlFormat` is `anthropic`. Default: `true`. |
| `supportsLongCacheRetention` | Whether the provider accepts long cache retention when cache retention is `long`: `prompt_cache_options.ttl: "30m"` for GPT-5.6+ Responses models, `prompt_cache_retention: "24h"` for earlier OpenAI models, or `cache_control.ttl: "1h"` when `cacheControlFormat` is `anthropic`. Default: `true`. |
| `openRouterRouting` | OpenRouter provider routing preferences. This object is sent as-is in the `provider` field of the [OpenRouter API request](https://openrouter.ai/docs/guides/routing/provider-selection). |
| `vercelGatewayRouting` | Vercel AI Gateway routing config for provider selection (`only`, `order`) |