mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-10-03 16:02:12 +00:00
fix: stop cutting DeepSeek chat turns off mid-thought (#11282)
* fix: stop cutting DeepSeek chat turns off mid-thought Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * test: pin the Anthropic max_tokens stop as a turn failure Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * fix: drop the send-failure prefix from the output-limit toast Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
46206a2787
commit
58217006ec
@@ -110,6 +110,7 @@ import type { Selection } from 'monaco-editor'
|
||||
import type AIChatInput from './AIChatInput.svelte'
|
||||
import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core'
|
||||
import { closeInterruptedToolBatch, runChatLoop, truncateToToolPairedPrefix } from './chatLoop'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
import { sanitizeToolCallArguments } from './toolCallArguments'
|
||||
import { billedTokens, normalizeContextUsage, type ChatTokenUsage } from './tokenUsage'
|
||||
import { logAiUsage } from '$lib/utils/aiUsageReporter'
|
||||
@@ -375,6 +376,10 @@ function isImageRejection(err: unknown, models: (string | undefined)[] = []): bo
|
||||
}
|
||||
|
||||
function getSendRequestErrorMessage(err: unknown, webSearchUnavailable: boolean): string {
|
||||
// The request went through; only its response was cut short.
|
||||
if (err instanceof OutputTokenLimitError) {
|
||||
return err.message
|
||||
}
|
||||
const errorMessage =
|
||||
err instanceof Error ? err.message : typeof err === 'string' ? err : undefined
|
||||
const message = errorMessage
|
||||
|
||||
@@ -1,6 +1,11 @@
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
import type { ChatCompletionMessageParam } from 'openai/resources/index.mjs'
|
||||
import { convertOpenAIToAnthropicMessages, partialWebSearchQuery } from './anthropic'
|
||||
import {
|
||||
convertOpenAIToAnthropicMessages,
|
||||
parseAnthropicCompletion,
|
||||
partialWebSearchQuery
|
||||
} from './anthropic'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
|
||||
// anthropic.ts pulls in the chat client/registry layer at import time; the
|
||||
// converter under test is pure, so stub those side-effecting modules away.
|
||||
@@ -217,3 +222,28 @@ describe('partialWebSearchQuery', () => {
|
||||
expect(partialWebSearchQuery('{"query": ')).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('parseAnthropicCompletion output token limit', () => {
|
||||
it('fails a message that stopped at max_tokens instead of ending the turn', async () => {
|
||||
const stream = {
|
||||
on: () => undefined,
|
||||
done: async () => undefined,
|
||||
finalMessage: async () => ({
|
||||
content: [{ type: 'thinking', thinking: 'I need to plan the flow so each' }],
|
||||
stop_reason: 'max_tokens',
|
||||
usage: { input_tokens: 20000, output_tokens: 64000 }
|
||||
})
|
||||
}
|
||||
|
||||
const parsed = parseAnthropicCompletion(
|
||||
stream as any,
|
||||
{ onNewToken: vi.fn(), onMessageEnd: vi.fn(), setToolStatus: vi.fn() } as any,
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
{}
|
||||
)
|
||||
|
||||
await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -24,6 +24,7 @@ import {
|
||||
type ToolCallbacks,
|
||||
type WebSearchSource
|
||||
} from './shared'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
import { anthropicUsageToChatTokenUsage, type ChatTokenUsage } from './tokenUsage'
|
||||
import { parseImageDataUrl } from './imageUtils'
|
||||
|
||||
@@ -453,6 +454,9 @@ export async function parseAnthropicCompletion(
|
||||
return { shouldContinue: true, tokenUsage }
|
||||
}
|
||||
|
||||
if (finalMessage.stop_reason === 'max_tokens') {
|
||||
throw new OutputTokenLimitError()
|
||||
}
|
||||
return { shouldContinue: false, tokenUsage }
|
||||
}
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
type ChatLoopConfig
|
||||
} from './chatLoop'
|
||||
import type { ReasoningProviderModel } from '../reasoningRegistry'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
|
||||
const mocks = vi.hoisted(() => ({
|
||||
getCompletion: vi.fn(),
|
||||
@@ -540,6 +541,23 @@ describe('runChatLoop lastIterationUsage', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('runChatLoop output token limit', () => {
|
||||
beforeEach(() => {
|
||||
vi.resetAllMocks()
|
||||
mocks.resolveRequestReasoning.mockReturnValue(undefined)
|
||||
})
|
||||
|
||||
it('fails a truncated Responses iteration instead of replaying it on the Completions API', async () => {
|
||||
mocks.getOpenAIResponsesCompletion.mockResolvedValue({})
|
||||
mocks.parseOpenAIResponsesCompletion.mockRejectedValue(new OutputTokenLimitError())
|
||||
|
||||
await expect(
|
||||
runChatLoop(createConfig({ workspace: `workspace-${randomUUID()}` }))
|
||||
).rejects.toBeInstanceOf(OutputTokenLimitError)
|
||||
expect(mocks.getCompletion).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
// Builders for the message shapes the chat loop accumulates.
|
||||
const assistant = (content: string): ChatCompletionMessageParam => ({ role: 'assistant', content })
|
||||
const assistantTools = (...ids: string[]): ChatCompletionMessageParam => ({
|
||||
|
||||
@@ -20,6 +20,7 @@ import {
|
||||
parseOpenAIResponsesCompletion
|
||||
} from './openai-responses'
|
||||
import type { Tool, ToolCallbacks } from './shared'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
import { sanitizeToolCallArguments } from './toolCallArguments'
|
||||
import { addChatTokenUsage, emptyChatTokenUsage, type ChatTokenUsage } from './tokenUsage'
|
||||
|
||||
@@ -501,6 +502,11 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
try {
|
||||
outcome = (await runOpenAIResponses(useWebSearch)) ? 'continue' : 'break'
|
||||
} catch (err) {
|
||||
// The response streamed and was then cut off: retrying it on the
|
||||
// Completions API would generate the whole iteration a second time.
|
||||
if (err instanceof OutputTokenLimitError) {
|
||||
throw err
|
||||
}
|
||||
if (reasoningSummary && shouldRetryWithoutReasoningSummary(err)) {
|
||||
markReasoningSummaryUnsupported(
|
||||
reasoningSummaryCacheKey,
|
||||
|
||||
@@ -3,8 +3,10 @@ import {
|
||||
buildPromptCacheKey,
|
||||
getOpenAIResponsesCompletion,
|
||||
openAIWebSearchDetails,
|
||||
parseOpenAIResponsesCompletion,
|
||||
toResponsesContent
|
||||
} from './openai-responses'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
|
||||
const mocks = vi.hoisted(() => ({
|
||||
getProviderAndCompletionConfig: vi.fn(),
|
||||
@@ -153,3 +155,34 @@ describe('openAIWebSearchDetails', () => {
|
||||
).toEqual({ query: undefined, sources: [{ url: 'https://a.dev' }] })
|
||||
})
|
||||
})
|
||||
|
||||
describe('parseOpenAIResponsesCompletion output token limit', () => {
|
||||
it('fails a response the stream reports incomplete at max_output_tokens', async () => {
|
||||
const handlers: Record<string, (event: any) => void> = {}
|
||||
// The SDK's final snapshot of an incomplete response still reads in_progress;
|
||||
// only the response.incomplete event carries the reason.
|
||||
const runner = {
|
||||
on: (name: string, fn: (event: any) => void) => {
|
||||
handlers[name] = fn
|
||||
},
|
||||
done: async () => {
|
||||
handlers['response.incomplete']?.({
|
||||
type: 'response.incomplete',
|
||||
response: { incomplete_details: { reason: 'max_output_tokens' } }
|
||||
})
|
||||
},
|
||||
finalResponse: async () => ({ status: 'in_progress', output: [] })
|
||||
}
|
||||
|
||||
const parsed = parseOpenAIResponsesCompletion(
|
||||
runner as any,
|
||||
{ onNewToken: vi.fn(), onMessageEnd: vi.fn(), setToolStatus: vi.fn() } as any,
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
{}
|
||||
)
|
||||
|
||||
await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -21,6 +21,7 @@ import {
|
||||
type ToolCallbacks,
|
||||
type WebSearchSource
|
||||
} from './shared'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
import type { ResponseStream } from 'openai/lib/responses/ResponseStream.mjs'
|
||||
import type { AIProviderModel } from '$lib/gen'
|
||||
import { openAIResponsesUsageToChatTokenUsage, type ChatTokenUsage } from './tokenUsage'
|
||||
@@ -542,6 +543,13 @@ export async function parseOpenAIResponsesCompletion(
|
||||
})
|
||||
})
|
||||
|
||||
// The stream's final snapshot keeps the in-progress status, so an
|
||||
// incomplete response is only visible through its own event.
|
||||
let hitOutputTokenLimit = false
|
||||
runner.on('response.incomplete', (event) => {
|
||||
hitOutputTokenLimit = event.response.incomplete_details?.reason === 'max_output_tokens'
|
||||
})
|
||||
|
||||
// Handle errors
|
||||
runner.on('error', (err: OpenAIError | ResponseErrorEvent) => {
|
||||
currentStreamingTool = undefined
|
||||
@@ -600,6 +608,9 @@ export async function parseOpenAIResponsesCompletion(
|
||||
return { shouldContinue: true, tokenUsage }
|
||||
}
|
||||
|
||||
if (hitOutputTokenLimit) {
|
||||
throw new OutputTokenLimitError()
|
||||
}
|
||||
return { shouldContinue: false, tokenUsage }
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
/** The provider ended the response at the request's output token cap. A turn
|
||||
* cut off there must fail rather than end quietly: thinking counts toward the
|
||||
* cap, so the stop often lands mid-thought and would read as the model giving
|
||||
* up. Failing also keeps the turn's output so far for a follow-up. */
|
||||
export class OutputTokenLimitError extends Error {
|
||||
constructor() {
|
||||
super(
|
||||
"The response was cut off at the model's output token limit. Ask it to continue, or raise the limit in the workspace AI settings under Model output limits."
|
||||
)
|
||||
this.name = 'OutputTokenLimitError'
|
||||
}
|
||||
}
|
||||
@@ -263,6 +263,7 @@ describe('model context windows', () => {
|
||||
expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('gemini-2.5-flash')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-v4-pro')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-flash')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-chat')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-reasoner')).toBe(1000000)
|
||||
})
|
||||
|
||||
@@ -285,3 +285,33 @@ describe('parseOpenAICompletion tool call arguments', () => {
|
||||
expect(toolResult.content).toBe('tool ok')
|
||||
})
|
||||
})
|
||||
|
||||
describe('output token limit', () => {
|
||||
it("gives DeepSeek DeepSeek's own thinking-mode budget rather than the 8192 fallback", async () => {
|
||||
const { getModelMaxTokens } = await import('./lib')
|
||||
expect(getModelMaxTokens('deepseek', 'deepseek-flash')).toBe(131072)
|
||||
expect(getModelMaxTokens('deepseek', 'deepseek-v4-pro')).toBe(131072)
|
||||
})
|
||||
|
||||
it('fails a response cut off at the output token limit instead of ending the turn', async () => {
|
||||
const { parseOpenAICompletion } = await import('./lib')
|
||||
const { OutputTokenLimitError } = await import('./chat/outputTokenLimit')
|
||||
|
||||
// Thinking counts toward max_tokens, so the cap can land before any answer.
|
||||
const parsed = parseOpenAICompletion(
|
||||
streamOf([
|
||||
{ choices: [{ delta: { reasoning_content: 'I need to plan the flow so each' } }] },
|
||||
{ choices: [{ delta: {}, finish_reason: 'length' }] }
|
||||
]),
|
||||
createCallbacks(),
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
{},
|
||||
undefined,
|
||||
{ workspace: 'test' }
|
||||
)
|
||||
|
||||
await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -28,6 +28,7 @@ import {
|
||||
type Tool,
|
||||
type ToolCallbacks
|
||||
} from './chat/shared'
|
||||
import { OutputTokenLimitError } from './chat/outputTokenLimit'
|
||||
import { hasValidToolCallArguments } from './chat/toolCallArguments'
|
||||
import {
|
||||
getNonStreamingOpenAIResponsesCompletion,
|
||||
@@ -116,7 +117,7 @@ export const AI_PROVIDERS: Record<AIProvider, AIProviderDetails> = {
|
||||
},
|
||||
deepseek: {
|
||||
label: 'DeepSeek',
|
||||
defaultModels: ['deepseek-v4-pro', 'deepseek-v4-flash']
|
||||
defaultModels: ['deepseek-v4-pro', 'deepseek-flash']
|
||||
},
|
||||
groq: {
|
||||
label: 'Groq',
|
||||
@@ -310,7 +311,12 @@ export async function fetchAvailableModels(
|
||||
}
|
||||
|
||||
export function getModelMaxTokens(provider: AIProvider, model: string) {
|
||||
if (model.includes('gpt-5')) {
|
||||
if (provider === 'deepseek') {
|
||||
// DeepSeek thinks by default and counts the thinking toward max_tokens, so
|
||||
// the 8192 fallback cuts long turns off mid-thought. 131072 is DeepSeek's
|
||||
// own default at the highest effort (65536 at the default one).
|
||||
return 131072
|
||||
} else if (model.includes('gpt-5')) {
|
||||
return 128000
|
||||
} else if (
|
||||
(provider === 'azure_openai' || provider === 'openai' || provider === 'azure_foundry') &&
|
||||
@@ -1218,6 +1224,7 @@ export async function parseOpenAICompletion(
|
||||
// to the next call, the previous one is demoted to queued.
|
||||
let streamingToolCallId: string | undefined = undefined
|
||||
let malformedFunctionCallError = false
|
||||
let hitOutputTokenLimit = false
|
||||
let tokenUsage = emptyChatTokenUsage()
|
||||
|
||||
let answer = ''
|
||||
@@ -1244,6 +1251,9 @@ export async function parseOpenAICompletion(
|
||||
) {
|
||||
malformedFunctionCallError = true
|
||||
}
|
||||
if (finishReason === 'length') {
|
||||
hitOutputTokenLimit = true
|
||||
}
|
||||
|
||||
// Mistral nests reasoning inside structured content parts; split them out
|
||||
// so a content delta never leaks "[object Object]" into the answer.
|
||||
@@ -1460,6 +1470,8 @@ export async function parseOpenAICompletion(
|
||||
}
|
||||
messages.push(toolResponse)
|
||||
addedMessages.push(toolResponse)
|
||||
} else if (hitOutputTokenLimit) {
|
||||
throw new OutputTokenLimitError()
|
||||
} else {
|
||||
return { shouldContinue: false, tokenUsage }
|
||||
}
|
||||
|
||||
@@ -96,10 +96,12 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
|
||||
['gemini-3.1', 1_000_000],
|
||||
['gemini-3', 1_000_000],
|
||||
['gemini-2.5', 1_000_000],
|
||||
// DeepSeek — the V4 family (pro / flash) is 1M. The deepseek-chat /
|
||||
// deepseek-reasoner aliases were retired 2026-07-24 but can still sit in a
|
||||
// saved selection, so they keep resolving to the window they had.
|
||||
// DeepSeek — the V4 family (pro / flash) is 1M; V4.1 Flash is served as
|
||||
// `deepseek-flash`. The deepseek-chat / deepseek-reasoner aliases were retired
|
||||
// 2026-07-24 but can still sit in a saved selection, so they keep resolving to
|
||||
// the window they had.
|
||||
['deepseek-v4', 1_000_000],
|
||||
['deepseek-flash', 1_000_000],
|
||||
['deepseek-chat', 1_000_000],
|
||||
['deepseek-reasoner', 1_000_000],
|
||||
['deepseek', 128_000],
|
||||
|
||||
Reference in New Issue
Block a user