fix: retry custom models without temperature

This commit is contained in:
nash_su
2026-08-20 14:11:23 +08:00
parent fd9c953551
commit 3df2f73c71
2 changed files with 57 additions and 0 deletions
+34
View File
@@ -130,6 +130,40 @@ describe("streamChat — buffered streaming responses", () => {
expect(onDone).not.toHaveBeenCalled()
})
it("retries a custom endpoint without temperature when the provider rejects it", async () => {
mockHttpFetch
.mockResolvedValueOnce(new Response(JSON.stringify({
error: { message: "Unsupported parameter: temperature" },
}), { status: 400 }))
.mockResolvedValueOnce(new Response([
openAiSseToken("retried"),
"data: [DONE]",
].join("\n\n"), { status: 200 }))
const onToken = vi.fn()
const onDone = vi.fn()
const onError = vi.fn()
await streamChat(
customStreamingCfg,
[{ role: "user", content: "hi" }],
{ onToken, onDone, onError },
undefined,
{ temperature: 0.1, max_tokens: 512 },
)
expect(mockHttpFetch).toHaveBeenCalledTimes(2)
expect(JSON.parse(String(mockHttpFetch.mock.calls[0][1]?.body))).toMatchObject({
temperature: 0.1,
max_tokens: 512,
})
const retryBody = JSON.parse(String(mockHttpFetch.mock.calls[1][1]?.body))
expect(retryBody.temperature).toBeUndefined()
expect(retryBody.max_tokens).toBe(512)
expect(onToken).toHaveBeenCalledWith("retried")
expect(onDone).toHaveBeenCalledTimes(1)
expect(onError).not.toHaveBeenCalled()
})
it("cancels a still-open response body after an SSE endpoint error", async () => {
let bodyCancelled = false
const body = new ReadableStream<Uint8Array>({
+23
View File
@@ -142,6 +142,25 @@ export function isReasoningOnlyResponseError(err: unknown): boolean {
return /^Model produced [\d,]+ characters of reasoning \/ chain-of-thought, but no actual response content\./.test(message)
}
function shouldRetryWithoutTemperature(
config: LlmConfig,
status: number,
errorDetail: string,
requestOverrides?: RequestOverrides,
): boolean {
if (config.provider !== "custom" || requestOverrides?.temperature === undefined) return false
if (status !== 400 && status !== 422) return false
const detail = errorDetail.toLowerCase()
return detail.includes("temperature") && (
detail.includes("unsupported") ||
detail.includes("not support") ||
detail.includes("unknown") ||
detail.includes("not allowed") ||
detail.includes("only") ||
detail.includes("invalid")
)
}
export async function streamChat(
config: LlmConfig,
messages: import("./llm-providers").ChatMessage[],
@@ -261,6 +280,10 @@ export async function streamChat(
} catch {
// ignore body read failure
}
if (shouldRetryWithoutTemperature(config, response.status, errorDetail, requestOverrides)) {
const { temperature: _temperature, ...retryOverrides } = requestOverrides ?? {}
return streamChat(config, messages, callbacks, signal, retryOverrides)
}
if (
response.status === 404 &&
(config.provider === "azure" ||