From 58217006ece56bbaf0447af67cac487e72b928c7 Mon Sep 17 00:00:00 2001 From: Ruben Fiszel Date: Tue, 22 Sep 2026 10:07:50 +0200 Subject: [PATCH] fix: stop cutting DeepSeek chat turns off mid-thought (#11282) * fix: stop cutting DeepSeek chat turns off mid-thought Co-Authored-By: Claude Opus 5 (1M context) * test: pin the Anthropic max_tokens stop as a turn failure Co-Authored-By: Claude Opus 5 (1M context) * fix: drop the send-failure prefix from the output-limit toast Co-Authored-By: Claude Opus 5 (1M context) --------- Co-authored-by: Claude Opus 5 (1M context) --- .../copilot/chat/AIChatManager.svelte.ts | 5 +++ .../components/copilot/chat/anthropic.test.ts | 32 +++++++++++++++++- .../lib/components/copilot/chat/anthropic.ts | 4 +++ .../components/copilot/chat/chatLoop.test.ts | 18 ++++++++++ .../lib/components/copilot/chat/chatLoop.ts | 6 ++++ .../copilot/chat/openai-responses.test.ts | 33 +++++++++++++++++++ .../copilot/chat/openai-responses.ts | 11 +++++++ .../copilot/chat/outputTokenLimit.ts | 12 +++++++ .../src/lib/components/copilot/lib.test.ts | 1 + .../components/copilot/lib.toolCalls.test.ts | 30 +++++++++++++++++ frontend/src/lib/components/copilot/lib.ts | 16 +++++++-- .../src/lib/components/copilot/modelConfig.ts | 8 +++-- 12 files changed, 170 insertions(+), 6 deletions(-) create mode 100644 frontend/src/lib/components/copilot/chat/outputTokenLimit.ts diff --git a/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts b/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts index c39325c84c..871dead4bc 100644 --- a/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts +++ b/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts @@ -110,6 +110,7 @@ import type { Selection } from 'monaco-editor' import type AIChatInput from './AIChatInput.svelte' import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core' import { closeInterruptedToolBatch, runChatLoop, truncateToToolPairedPrefix } from './chatLoop' +import { OutputTokenLimitError } from './outputTokenLimit' import { sanitizeToolCallArguments } from './toolCallArguments' import { billedTokens, normalizeContextUsage, type ChatTokenUsage } from './tokenUsage' import { logAiUsage } from '$lib/utils/aiUsageReporter' @@ -375,6 +376,10 @@ function isImageRejection(err: unknown, models: (string | undefined)[] = []): bo } function getSendRequestErrorMessage(err: unknown, webSearchUnavailable: boolean): string { + // The request went through; only its response was cut short. + if (err instanceof OutputTokenLimitError) { + return err.message + } const errorMessage = err instanceof Error ? err.message : typeof err === 'string' ? err : undefined const message = errorMessage diff --git a/frontend/src/lib/components/copilot/chat/anthropic.test.ts b/frontend/src/lib/components/copilot/chat/anthropic.test.ts index 1ac413973f..9f6b49a9e9 100644 --- a/frontend/src/lib/components/copilot/chat/anthropic.test.ts +++ b/frontend/src/lib/components/copilot/chat/anthropic.test.ts @@ -1,6 +1,11 @@ import { describe, expect, it, vi } from 'vitest' import type { ChatCompletionMessageParam } from 'openai/resources/index.mjs' -import { convertOpenAIToAnthropicMessages, partialWebSearchQuery } from './anthropic' +import { + convertOpenAIToAnthropicMessages, + parseAnthropicCompletion, + partialWebSearchQuery +} from './anthropic' +import { OutputTokenLimitError } from './outputTokenLimit' // anthropic.ts pulls in the chat client/registry layer at import time; the // converter under test is pure, so stub those side-effecting modules away. @@ -217,3 +222,28 @@ describe('partialWebSearchQuery', () => { expect(partialWebSearchQuery('{"query": ')).toBeUndefined() }) }) + +describe('parseAnthropicCompletion output token limit', () => { + it('fails a message that stopped at max_tokens instead of ending the turn', async () => { + const stream = { + on: () => undefined, + done: async () => undefined, + finalMessage: async () => ({ + content: [{ type: 'thinking', thinking: 'I need to plan the flow so each' }], + stop_reason: 'max_tokens', + usage: { input_tokens: 20000, output_tokens: 64000 } + }) + } + + const parsed = parseAnthropicCompletion( + stream as any, + { onNewToken: vi.fn(), onMessageEnd: vi.fn(), setToolStatus: vi.fn() } as any, + [], + [], + [], + {} + ) + + await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError) + }) +}) diff --git a/frontend/src/lib/components/copilot/chat/anthropic.ts b/frontend/src/lib/components/copilot/chat/anthropic.ts index 8d38f10916..a311d16990 100644 --- a/frontend/src/lib/components/copilot/chat/anthropic.ts +++ b/frontend/src/lib/components/copilot/chat/anthropic.ts @@ -24,6 +24,7 @@ import { type ToolCallbacks, type WebSearchSource } from './shared' +import { OutputTokenLimitError } from './outputTokenLimit' import { anthropicUsageToChatTokenUsage, type ChatTokenUsage } from './tokenUsage' import { parseImageDataUrl } from './imageUtils' @@ -453,6 +454,9 @@ export async function parseAnthropicCompletion( return { shouldContinue: true, tokenUsage } } + if (finalMessage.stop_reason === 'max_tokens') { + throw new OutputTokenLimitError() + } return { shouldContinue: false, tokenUsage } } diff --git a/frontend/src/lib/components/copilot/chat/chatLoop.test.ts b/frontend/src/lib/components/copilot/chat/chatLoop.test.ts index 251c2ff858..144617779f 100644 --- a/frontend/src/lib/components/copilot/chat/chatLoop.test.ts +++ b/frontend/src/lib/components/copilot/chat/chatLoop.test.ts @@ -8,6 +8,7 @@ import { type ChatLoopConfig } from './chatLoop' import type { ReasoningProviderModel } from '../reasoningRegistry' +import { OutputTokenLimitError } from './outputTokenLimit' const mocks = vi.hoisted(() => ({ getCompletion: vi.fn(), @@ -540,6 +541,23 @@ describe('runChatLoop lastIterationUsage', () => { }) }) +describe('runChatLoop output token limit', () => { + beforeEach(() => { + vi.resetAllMocks() + mocks.resolveRequestReasoning.mockReturnValue(undefined) + }) + + it('fails a truncated Responses iteration instead of replaying it on the Completions API', async () => { + mocks.getOpenAIResponsesCompletion.mockResolvedValue({}) + mocks.parseOpenAIResponsesCompletion.mockRejectedValue(new OutputTokenLimitError()) + + await expect( + runChatLoop(createConfig({ workspace: `workspace-${randomUUID()}` })) + ).rejects.toBeInstanceOf(OutputTokenLimitError) + expect(mocks.getCompletion).not.toHaveBeenCalled() + }) +}) + // Builders for the message shapes the chat loop accumulates. const assistant = (content: string): ChatCompletionMessageParam => ({ role: 'assistant', content }) const assistantTools = (...ids: string[]): ChatCompletionMessageParam => ({ diff --git a/frontend/src/lib/components/copilot/chat/chatLoop.ts b/frontend/src/lib/components/copilot/chat/chatLoop.ts index 110e4f1154..5091d76af5 100644 --- a/frontend/src/lib/components/copilot/chat/chatLoop.ts +++ b/frontend/src/lib/components/copilot/chat/chatLoop.ts @@ -20,6 +20,7 @@ import { parseOpenAIResponsesCompletion } from './openai-responses' import type { Tool, ToolCallbacks } from './shared' +import { OutputTokenLimitError } from './outputTokenLimit' import { sanitizeToolCallArguments } from './toolCallArguments' import { addChatTokenUsage, emptyChatTokenUsage, type ChatTokenUsage } from './tokenUsage' @@ -501,6 +502,11 @@ export async function runChatLoop(config: ChatLoopConfig): Promise ({ getProviderAndCompletionConfig: vi.fn(), @@ -153,3 +155,34 @@ describe('openAIWebSearchDetails', () => { ).toEqual({ query: undefined, sources: [{ url: 'https://a.dev' }] }) }) }) + +describe('parseOpenAIResponsesCompletion output token limit', () => { + it('fails a response the stream reports incomplete at max_output_tokens', async () => { + const handlers: Record void> = {} + // The SDK's final snapshot of an incomplete response still reads in_progress; + // only the response.incomplete event carries the reason. + const runner = { + on: (name: string, fn: (event: any) => void) => { + handlers[name] = fn + }, + done: async () => { + handlers['response.incomplete']?.({ + type: 'response.incomplete', + response: { incomplete_details: { reason: 'max_output_tokens' } } + }) + }, + finalResponse: async () => ({ status: 'in_progress', output: [] }) + } + + const parsed = parseOpenAIResponsesCompletion( + runner as any, + { onNewToken: vi.fn(), onMessageEnd: vi.fn(), setToolStatus: vi.fn() } as any, + [], + [], + [], + {} + ) + + await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError) + }) +}) diff --git a/frontend/src/lib/components/copilot/chat/openai-responses.ts b/frontend/src/lib/components/copilot/chat/openai-responses.ts index 4118eb492d..b216a08107 100644 --- a/frontend/src/lib/components/copilot/chat/openai-responses.ts +++ b/frontend/src/lib/components/copilot/chat/openai-responses.ts @@ -21,6 +21,7 @@ import { type ToolCallbacks, type WebSearchSource } from './shared' +import { OutputTokenLimitError } from './outputTokenLimit' import type { ResponseStream } from 'openai/lib/responses/ResponseStream.mjs' import type { AIProviderModel } from '$lib/gen' import { openAIResponsesUsageToChatTokenUsage, type ChatTokenUsage } from './tokenUsage' @@ -542,6 +543,13 @@ export async function parseOpenAIResponsesCompletion( }) }) + // The stream's final snapshot keeps the in-progress status, so an + // incomplete response is only visible through its own event. + let hitOutputTokenLimit = false + runner.on('response.incomplete', (event) => { + hitOutputTokenLimit = event.response.incomplete_details?.reason === 'max_output_tokens' + }) + // Handle errors runner.on('error', (err: OpenAIError | ResponseErrorEvent) => { currentStreamingTool = undefined @@ -600,6 +608,9 @@ export async function parseOpenAIResponsesCompletion( return { shouldContinue: true, tokenUsage } } + if (hitOutputTokenLimit) { + throw new OutputTokenLimitError() + } return { shouldContinue: false, tokenUsage } } diff --git a/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts b/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts new file mode 100644 index 0000000000..b1127a1db7 --- /dev/null +++ b/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts @@ -0,0 +1,12 @@ +/** The provider ended the response at the request's output token cap. A turn + * cut off there must fail rather than end quietly: thinking counts toward the + * cap, so the stop often lands mid-thought and would read as the model giving + * up. Failing also keeps the turn's output so far for a follow-up. */ +export class OutputTokenLimitError extends Error { + constructor() { + super( + "The response was cut off at the model's output token limit. Ask it to continue, or raise the limit in the workspace AI settings under Model output limits." + ) + this.name = 'OutputTokenLimitError' + } +} diff --git a/frontend/src/lib/components/copilot/lib.test.ts b/frontend/src/lib/components/copilot/lib.test.ts index 5f1c657b39..8e06e0e2b9 100644 --- a/frontend/src/lib/components/copilot/lib.test.ts +++ b/frontend/src/lib/components/copilot/lib.test.ts @@ -263,6 +263,7 @@ describe('model context windows', () => { expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000) expect(getKnownModelContextWindow('gemini-2.5-flash')).toBe(1000000) expect(getKnownModelContextWindow('deepseek-v4-pro')).toBe(1000000) + expect(getKnownModelContextWindow('deepseek-flash')).toBe(1000000) expect(getKnownModelContextWindow('deepseek-chat')).toBe(1000000) expect(getKnownModelContextWindow('deepseek-reasoner')).toBe(1000000) }) diff --git a/frontend/src/lib/components/copilot/lib.toolCalls.test.ts b/frontend/src/lib/components/copilot/lib.toolCalls.test.ts index c665cc964e..b011c1f672 100644 --- a/frontend/src/lib/components/copilot/lib.toolCalls.test.ts +++ b/frontend/src/lib/components/copilot/lib.toolCalls.test.ts @@ -285,3 +285,33 @@ describe('parseOpenAICompletion tool call arguments', () => { expect(toolResult.content).toBe('tool ok') }) }) + +describe('output token limit', () => { + it("gives DeepSeek DeepSeek's own thinking-mode budget rather than the 8192 fallback", async () => { + const { getModelMaxTokens } = await import('./lib') + expect(getModelMaxTokens('deepseek', 'deepseek-flash')).toBe(131072) + expect(getModelMaxTokens('deepseek', 'deepseek-v4-pro')).toBe(131072) + }) + + it('fails a response cut off at the output token limit instead of ending the turn', async () => { + const { parseOpenAICompletion } = await import('./lib') + const { OutputTokenLimitError } = await import('./chat/outputTokenLimit') + + // Thinking counts toward max_tokens, so the cap can land before any answer. + const parsed = parseOpenAICompletion( + streamOf([ + { choices: [{ delta: { reasoning_content: 'I need to plan the flow so each' } }] }, + { choices: [{ delta: {}, finish_reason: 'length' }] } + ]), + createCallbacks(), + [], + [], + [], + {}, + undefined, + { workspace: 'test' } + ) + + await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError) + }) +}) diff --git a/frontend/src/lib/components/copilot/lib.ts b/frontend/src/lib/components/copilot/lib.ts index 19f350a665..5094764807 100644 --- a/frontend/src/lib/components/copilot/lib.ts +++ b/frontend/src/lib/components/copilot/lib.ts @@ -28,6 +28,7 @@ import { type Tool, type ToolCallbacks } from './chat/shared' +import { OutputTokenLimitError } from './chat/outputTokenLimit' import { hasValidToolCallArguments } from './chat/toolCallArguments' import { getNonStreamingOpenAIResponsesCompletion, @@ -116,7 +117,7 @@ export const AI_PROVIDERS: Record = { }, deepseek: { label: 'DeepSeek', - defaultModels: ['deepseek-v4-pro', 'deepseek-v4-flash'] + defaultModels: ['deepseek-v4-pro', 'deepseek-flash'] }, groq: { label: 'Groq', @@ -310,7 +311,12 @@ export async function fetchAvailableModels( } export function getModelMaxTokens(provider: AIProvider, model: string) { - if (model.includes('gpt-5')) { + if (provider === 'deepseek') { + // DeepSeek thinks by default and counts the thinking toward max_tokens, so + // the 8192 fallback cuts long turns off mid-thought. 131072 is DeepSeek's + // own default at the highest effort (65536 at the default one). + return 131072 + } else if (model.includes('gpt-5')) { return 128000 } else if ( (provider === 'azure_openai' || provider === 'openai' || provider === 'azure_foundry') && @@ -1218,6 +1224,7 @@ export async function parseOpenAICompletion( // to the next call, the previous one is demoted to queued. let streamingToolCallId: string | undefined = undefined let malformedFunctionCallError = false + let hitOutputTokenLimit = false let tokenUsage = emptyChatTokenUsage() let answer = '' @@ -1244,6 +1251,9 @@ export async function parseOpenAICompletion( ) { malformedFunctionCallError = true } + if (finishReason === 'length') { + hitOutputTokenLimit = true + } // Mistral nests reasoning inside structured content parts; split them out // so a content delta never leaks "[object Object]" into the answer. @@ -1460,6 +1470,8 @@ export async function parseOpenAICompletion( } messages.push(toolResponse) addedMessages.push(toolResponse) + } else if (hitOutputTokenLimit) { + throw new OutputTokenLimitError() } else { return { shouldContinue: false, tokenUsage } } diff --git a/frontend/src/lib/components/copilot/modelConfig.ts b/frontend/src/lib/components/copilot/modelConfig.ts index e5a66248b3..be49907ed4 100644 --- a/frontend/src/lib/components/copilot/modelConfig.ts +++ b/frontend/src/lib/components/copilot/modelConfig.ts @@ -96,10 +96,12 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [ ['gemini-3.1', 1_000_000], ['gemini-3', 1_000_000], ['gemini-2.5', 1_000_000], - // DeepSeek — the V4 family (pro / flash) is 1M. The deepseek-chat / - // deepseek-reasoner aliases were retired 2026-07-24 but can still sit in a - // saved selection, so they keep resolving to the window they had. + // DeepSeek — the V4 family (pro / flash) is 1M; V4.1 Flash is served as + // `deepseek-flash`. The deepseek-chat / deepseek-reasoner aliases were retired + // 2026-07-24 but can still sit in a saved selection, so they keep resolving to + // the window they had. ['deepseek-v4', 1_000_000], + ['deepseek-flash', 1_000_000], ['deepseek-chat', 1_000_000], ['deepseek-reasoner', 1_000_000], ['deepseek', 128_000],