fix: stop cutting DeepSeek chat turns off mid-thought (#11282)

* fix: stop cutting DeepSeek chat turns off mid-thought

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* test: pin the Anthropic max_tokens stop as a turn failure

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* fix: drop the send-failure prefix from the output-limit toast

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Ruben Fiszel
2026-09-22 08:07:50 +00:00
committed by GitHub
co-authored by Claude Opus 5
parent 46206a2787
commit 58217006ec
12 changed files with 170 additions and 6 deletions
@@ -110,6 +110,7 @@ import type { Selection } from 'monaco-editor'
import type AIChatInput from './AIChatInput.svelte'
import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core'
import { closeInterruptedToolBatch, runChatLoop, truncateToToolPairedPrefix } from './chatLoop'
import { OutputTokenLimitError } from './outputTokenLimit'
import { sanitizeToolCallArguments } from './toolCallArguments'
import { billedTokens, normalizeContextUsage, type ChatTokenUsage } from './tokenUsage'
import { logAiUsage } from '$lib/utils/aiUsageReporter'
@@ -375,6 +376,10 @@ function isImageRejection(err: unknown, models: (string | undefined)[] = []): bo
}
function getSendRequestErrorMessage(err: unknown, webSearchUnavailable: boolean): string {
// The request went through; only its response was cut short.
if (err instanceof OutputTokenLimitError) {
return err.message
}
const errorMessage =
err instanceof Error ? err.message : typeof err === 'string' ? err : undefined
const message = errorMessage
@@ -1,6 +1,11 @@
import { describe, expect, it, vi } from 'vitest'
import type { ChatCompletionMessageParam } from 'openai/resources/index.mjs'
import { convertOpenAIToAnthropicMessages, partialWebSearchQuery } from './anthropic'
import {
convertOpenAIToAnthropicMessages,
parseAnthropicCompletion,
partialWebSearchQuery
} from './anthropic'
import { OutputTokenLimitError } from './outputTokenLimit'
// anthropic.ts pulls in the chat client/registry layer at import time; the
// converter under test is pure, so stub those side-effecting modules away.
@@ -217,3 +222,28 @@ describe('partialWebSearchQuery', () => {
expect(partialWebSearchQuery('{"query": ')).toBeUndefined()
})
})
describe('parseAnthropicCompletion output token limit', () => {
it('fails a message that stopped at max_tokens instead of ending the turn', async () => {
const stream = {
on: () => undefined,
done: async () => undefined,
finalMessage: async () => ({
content: [{ type: 'thinking', thinking: 'I need to plan the flow so each' }],
stop_reason: 'max_tokens',
usage: { input_tokens: 20000, output_tokens: 64000 }
})
}
const parsed = parseAnthropicCompletion(
stream as any,
{ onNewToken: vi.fn(), onMessageEnd: vi.fn(), setToolStatus: vi.fn() } as any,
[],
[],
[],
{}
)
await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError)
})
})
@@ -24,6 +24,7 @@ import {
type ToolCallbacks,
type WebSearchSource
} from './shared'
import { OutputTokenLimitError } from './outputTokenLimit'
import { anthropicUsageToChatTokenUsage, type ChatTokenUsage } from './tokenUsage'
import { parseImageDataUrl } from './imageUtils'
@@ -453,6 +454,9 @@ export async function parseAnthropicCompletion(
return { shouldContinue: true, tokenUsage }
}
if (finalMessage.stop_reason === 'max_tokens') {
throw new OutputTokenLimitError()
}
return { shouldContinue: false, tokenUsage }
}
@@ -8,6 +8,7 @@ import {
type ChatLoopConfig
} from './chatLoop'
import type { ReasoningProviderModel } from '../reasoningRegistry'
import { OutputTokenLimitError } from './outputTokenLimit'
const mocks = vi.hoisted(() => ({
getCompletion: vi.fn(),
@@ -540,6 +541,23 @@ describe('runChatLoop lastIterationUsage', () => {
})
})
describe('runChatLoop output token limit', () => {
beforeEach(() => {
vi.resetAllMocks()
mocks.resolveRequestReasoning.mockReturnValue(undefined)
})
it('fails a truncated Responses iteration instead of replaying it on the Completions API', async () => {
mocks.getOpenAIResponsesCompletion.mockResolvedValue({})
mocks.parseOpenAIResponsesCompletion.mockRejectedValue(new OutputTokenLimitError())
await expect(
runChatLoop(createConfig({ workspace: `workspace-${randomUUID()}` }))
).rejects.toBeInstanceOf(OutputTokenLimitError)
expect(mocks.getCompletion).not.toHaveBeenCalled()
})
})
// Builders for the message shapes the chat loop accumulates.
const assistant = (content: string): ChatCompletionMessageParam => ({ role: 'assistant', content })
const assistantTools = (...ids: string[]): ChatCompletionMessageParam => ({
@@ -20,6 +20,7 @@ import {
parseOpenAIResponsesCompletion
} from './openai-responses'
import type { Tool, ToolCallbacks } from './shared'
import { OutputTokenLimitError } from './outputTokenLimit'
import { sanitizeToolCallArguments } from './toolCallArguments'
import { addChatTokenUsage, emptyChatTokenUsage, type ChatTokenUsage } from './tokenUsage'
@@ -501,6 +502,11 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
try {
outcome = (await runOpenAIResponses(useWebSearch)) ? 'continue' : 'break'
} catch (err) {
// The response streamed and was then cut off: retrying it on the
// Completions API would generate the whole iteration a second time.
if (err instanceof OutputTokenLimitError) {
throw err
}
if (reasoningSummary && shouldRetryWithoutReasoningSummary(err)) {
markReasoningSummaryUnsupported(
reasoningSummaryCacheKey,
@@ -3,8 +3,10 @@ import {
buildPromptCacheKey,
getOpenAIResponsesCompletion,
openAIWebSearchDetails,
parseOpenAIResponsesCompletion,
toResponsesContent
} from './openai-responses'
import { OutputTokenLimitError } from './outputTokenLimit'
const mocks = vi.hoisted(() => ({
getProviderAndCompletionConfig: vi.fn(),
@@ -153,3 +155,34 @@ describe('openAIWebSearchDetails', () => {
).toEqual({ query: undefined, sources: [{ url: 'https://a.dev' }] })
})
})
describe('parseOpenAIResponsesCompletion output token limit', () => {
it('fails a response the stream reports incomplete at max_output_tokens', async () => {
const handlers: Record<string, (event: any) => void> = {}
// The SDK's final snapshot of an incomplete response still reads in_progress;
// only the response.incomplete event carries the reason.
const runner = {
on: (name: string, fn: (event: any) => void) => {
handlers[name] = fn
},
done: async () => {
handlers['response.incomplete']?.({
type: 'response.incomplete',
response: { incomplete_details: { reason: 'max_output_tokens' } }
})
},
finalResponse: async () => ({ status: 'in_progress', output: [] })
}
const parsed = parseOpenAIResponsesCompletion(
runner as any,
{ onNewToken: vi.fn(), onMessageEnd: vi.fn(), setToolStatus: vi.fn() } as any,
[],
[],
[],
{}
)
await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError)
})
})
@@ -21,6 +21,7 @@ import {
type ToolCallbacks,
type WebSearchSource
} from './shared'
import { OutputTokenLimitError } from './outputTokenLimit'
import type { ResponseStream } from 'openai/lib/responses/ResponseStream.mjs'
import type { AIProviderModel } from '$lib/gen'
import { openAIResponsesUsageToChatTokenUsage, type ChatTokenUsage } from './tokenUsage'
@@ -542,6 +543,13 @@ export async function parseOpenAIResponsesCompletion(
})
})
// The stream's final snapshot keeps the in-progress status, so an
// incomplete response is only visible through its own event.
let hitOutputTokenLimit = false
runner.on('response.incomplete', (event) => {
hitOutputTokenLimit = event.response.incomplete_details?.reason === 'max_output_tokens'
})
// Handle errors
runner.on('error', (err: OpenAIError | ResponseErrorEvent) => {
currentStreamingTool = undefined
@@ -600,6 +608,9 @@ export async function parseOpenAIResponsesCompletion(
return { shouldContinue: true, tokenUsage }
}
if (hitOutputTokenLimit) {
throw new OutputTokenLimitError()
}
return { shouldContinue: false, tokenUsage }
}
@@ -0,0 +1,12 @@
/** The provider ended the response at the request's output token cap. A turn
* cut off there must fail rather than end quietly: thinking counts toward the
* cap, so the stop often lands mid-thought and would read as the model giving
* up. Failing also keeps the turn's output so far for a follow-up. */
export class OutputTokenLimitError extends Error {
constructor() {
super(
"The response was cut off at the model's output token limit. Ask it to continue, or raise the limit in the workspace AI settings under Model output limits."
)
this.name = 'OutputTokenLimitError'
}
}
@@ -263,6 +263,7 @@ describe('model context windows', () => {
expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000)
expect(getKnownModelContextWindow('gemini-2.5-flash')).toBe(1000000)
expect(getKnownModelContextWindow('deepseek-v4-pro')).toBe(1000000)
expect(getKnownModelContextWindow('deepseek-flash')).toBe(1000000)
expect(getKnownModelContextWindow('deepseek-chat')).toBe(1000000)
expect(getKnownModelContextWindow('deepseek-reasoner')).toBe(1000000)
})
@@ -285,3 +285,33 @@ describe('parseOpenAICompletion tool call arguments', () => {
expect(toolResult.content).toBe('tool ok')
})
})
describe('output token limit', () => {
it("gives DeepSeek DeepSeek's own thinking-mode budget rather than the 8192 fallback", async () => {
const { getModelMaxTokens } = await import('./lib')
expect(getModelMaxTokens('deepseek', 'deepseek-flash')).toBe(131072)
expect(getModelMaxTokens('deepseek', 'deepseek-v4-pro')).toBe(131072)
})
it('fails a response cut off at the output token limit instead of ending the turn', async () => {
const { parseOpenAICompletion } = await import('./lib')
const { OutputTokenLimitError } = await import('./chat/outputTokenLimit')
// Thinking counts toward max_tokens, so the cap can land before any answer.
const parsed = parseOpenAICompletion(
streamOf([
{ choices: [{ delta: { reasoning_content: 'I need to plan the flow so each' } }] },
{ choices: [{ delta: {}, finish_reason: 'length' }] }
]),
createCallbacks(),
[],
[],
[],
{},
undefined,
{ workspace: 'test' }
)
await expect(parsed).rejects.toBeInstanceOf(OutputTokenLimitError)
})
})
+14 -2
View File
@@ -28,6 +28,7 @@ import {
type Tool,
type ToolCallbacks
} from './chat/shared'
import { OutputTokenLimitError } from './chat/outputTokenLimit'
import { hasValidToolCallArguments } from './chat/toolCallArguments'
import {
getNonStreamingOpenAIResponsesCompletion,
@@ -116,7 +117,7 @@ export const AI_PROVIDERS: Record<AIProvider, AIProviderDetails> = {
},
deepseek: {
label: 'DeepSeek',
defaultModels: ['deepseek-v4-pro', 'deepseek-v4-flash']
defaultModels: ['deepseek-v4-pro', 'deepseek-flash']
},
groq: {
label: 'Groq',
@@ -310,7 +311,12 @@ export async function fetchAvailableModels(
}
export function getModelMaxTokens(provider: AIProvider, model: string) {
if (model.includes('gpt-5')) {
if (provider === 'deepseek') {
// DeepSeek thinks by default and counts the thinking toward max_tokens, so
// the 8192 fallback cuts long turns off mid-thought. 131072 is DeepSeek's
// own default at the highest effort (65536 at the default one).
return 131072
} else if (model.includes('gpt-5')) {
return 128000
} else if (
(provider === 'azure_openai' || provider === 'openai' || provider === 'azure_foundry') &&
@@ -1218,6 +1224,7 @@ export async function parseOpenAICompletion(
// to the next call, the previous one is demoted to queued.
let streamingToolCallId: string | undefined = undefined
let malformedFunctionCallError = false
let hitOutputTokenLimit = false
let tokenUsage = emptyChatTokenUsage()
let answer = ''
@@ -1244,6 +1251,9 @@ export async function parseOpenAICompletion(
) {
malformedFunctionCallError = true
}
if (finishReason === 'length') {
hitOutputTokenLimit = true
}
// Mistral nests reasoning inside structured content parts; split them out
// so a content delta never leaks "[object Object]" into the answer.
@@ -1460,6 +1470,8 @@ export async function parseOpenAICompletion(
}
messages.push(toolResponse)
addedMessages.push(toolResponse)
} else if (hitOutputTokenLimit) {
throw new OutputTokenLimitError()
} else {
return { shouldContinue: false, tokenUsage }
}
@@ -96,10 +96,12 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
['gemini-3.1', 1_000_000],
['gemini-3', 1_000_000],
['gemini-2.5', 1_000_000],
// DeepSeek — the V4 family (pro / flash) is 1M. The deepseek-chat /
// deepseek-reasoner aliases were retired 2026-07-24 but can still sit in a
// saved selection, so they keep resolving to the window they had.
// DeepSeek — the V4 family (pro / flash) is 1M; V4.1 Flash is served as
// `deepseek-flash`. The deepseek-chat / deepseek-reasoner aliases were retired
// 2026-07-24 but can still sit in a saved selection, so they keep resolving to
// the window they had.
['deepseek-v4', 1_000_000],
['deepseek-flash', 1_000_000],
['deepseek-chat', 1_000_000],
['deepseek-reasoner', 1_000_000],
['deepseek', 128_000],