mirror of
https://github.com/nashsu/llm_wiki.git
synced 2026-10-02 02:44:34 +08:00
fix: retry custom models without temperature
This commit is contained in:
@@ -130,6 +130,40 @@ describe("streamChat — buffered streaming responses", () => {
|
||||
expect(onDone).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it("retries a custom endpoint without temperature when the provider rejects it", async () => {
|
||||
mockHttpFetch
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify({
|
||||
error: { message: "Unsupported parameter: temperature" },
|
||||
}), { status: 400 }))
|
||||
.mockResolvedValueOnce(new Response([
|
||||
openAiSseToken("retried"),
|
||||
"data: [DONE]",
|
||||
].join("\n\n"), { status: 200 }))
|
||||
const onToken = vi.fn()
|
||||
const onDone = vi.fn()
|
||||
const onError = vi.fn()
|
||||
|
||||
await streamChat(
|
||||
customStreamingCfg,
|
||||
[{ role: "user", content: "hi" }],
|
||||
{ onToken, onDone, onError },
|
||||
undefined,
|
||||
{ temperature: 0.1, max_tokens: 512 },
|
||||
)
|
||||
|
||||
expect(mockHttpFetch).toHaveBeenCalledTimes(2)
|
||||
expect(JSON.parse(String(mockHttpFetch.mock.calls[0][1]?.body))).toMatchObject({
|
||||
temperature: 0.1,
|
||||
max_tokens: 512,
|
||||
})
|
||||
const retryBody = JSON.parse(String(mockHttpFetch.mock.calls[1][1]?.body))
|
||||
expect(retryBody.temperature).toBeUndefined()
|
||||
expect(retryBody.max_tokens).toBe(512)
|
||||
expect(onToken).toHaveBeenCalledWith("retried")
|
||||
expect(onDone).toHaveBeenCalledTimes(1)
|
||||
expect(onError).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it("cancels a still-open response body after an SSE endpoint error", async () => {
|
||||
let bodyCancelled = false
|
||||
const body = new ReadableStream<Uint8Array>({
|
||||
|
||||
@@ -142,6 +142,25 @@ export function isReasoningOnlyResponseError(err: unknown): boolean {
|
||||
return /^Model produced [\d,]+ characters of reasoning \/ chain-of-thought, but no actual response content\./.test(message)
|
||||
}
|
||||
|
||||
function shouldRetryWithoutTemperature(
|
||||
config: LlmConfig,
|
||||
status: number,
|
||||
errorDetail: string,
|
||||
requestOverrides?: RequestOverrides,
|
||||
): boolean {
|
||||
if (config.provider !== "custom" || requestOverrides?.temperature === undefined) return false
|
||||
if (status !== 400 && status !== 422) return false
|
||||
const detail = errorDetail.toLowerCase()
|
||||
return detail.includes("temperature") && (
|
||||
detail.includes("unsupported") ||
|
||||
detail.includes("not support") ||
|
||||
detail.includes("unknown") ||
|
||||
detail.includes("not allowed") ||
|
||||
detail.includes("only") ||
|
||||
detail.includes("invalid")
|
||||
)
|
||||
}
|
||||
|
||||
export async function streamChat(
|
||||
config: LlmConfig,
|
||||
messages: import("./llm-providers").ChatMessage[],
|
||||
@@ -261,6 +280,10 @@ export async function streamChat(
|
||||
} catch {
|
||||
// ignore body read failure
|
||||
}
|
||||
if (shouldRetryWithoutTemperature(config, response.status, errorDetail, requestOverrides)) {
|
||||
const { temperature: _temperature, ...retryOverrides } = requestOverrides ?? {}
|
||||
return streamChat(config, messages, callbacks, signal, retryOverrides)
|
||||
}
|
||||
if (
|
||||
response.status === 404 &&
|
||||
(config.provider === "azure" ||
|
||||
|
||||
Reference in New Issue
Block a user