From 7d0f2ce3bf920a68f1ea22d59481d239947f1b51 Mon Sep 17 00:00:00 2001 From: Ruben Fiszel Date: Tue, 22 Sep 2026 11:32:11 +0200 Subject: [PATCH] fix: size the AI chat output budget by model, not provider (#11286) * fix: size the AI chat output budget by model, not provider Co-Authored-By: Claude Opus 5 (1M context) * fix: keep self-hosted open-weight models on the fallback output budget Co-Authored-By: Claude Opus 5 (1M context) * chore: point ee-repo-ref at the free tier output clamp change Co-Authored-By: Claude Opus 5 (1M context) * fix: treat codestral as open-weight for the output budget Co-Authored-By: Claude Opus 5 (1M context) * chore: update ee-repo-ref to 559a9ff78ea03439553cf2b412e73765762e9013 This commit updates the EE repository reference after PR #821 was merged in windmill-ee-private. Previous ee-repo-ref: 953f7c1cc9a70b740d5d1b129cb1401950854b02 New ee-repo-ref: 559a9ff78ea03439553cf2b412e73765762e9013 Automated by sync-ee-ref workflow. --------- Co-authored-by: Claude Opus 5 (1M context) Co-authored-by: windmill-internal-app[bot] --- backend/ee-repo-ref.txt | 2 +- .../copilot/chat/AIChatManager.svelte.ts | 4 +- .../lib/components/copilot/chat/anthropic.ts | 3 +- .../copilot/chat/openai-responses.ts | 6 +- .../copilot/chat/outputTokenLimit.ts | 5 + .../components/copilot/lib.toolCalls.test.ts | 36 ++++++- frontend/src/lib/components/copilot/lib.ts | 100 ++++++++++-------- .../src/lib/components/copilot/modelConfig.ts | 65 ++++++++++++ .../components/copilot/reasoningRegistry.ts | 17 ++- .../workspaceSettings/AISettings.svelte | 4 +- 10 files changed, 185 insertions(+), 57 deletions(-) diff --git a/backend/ee-repo-ref.txt b/backend/ee-repo-ref.txt index c336c4fdd3..b82e25d93b 100644 --- a/backend/ee-repo-ref.txt +++ b/backend/ee-repo-ref.txt @@ -1 +1 @@ -5370ae3a95a3dc10171f72da74286a654996eee6 +559a9ff78ea03439553cf2b412e73765762e9013 diff --git a/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts b/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts index 871dead4bc..27f9bfc43f 100644 --- a/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts +++ b/frontend/src/lib/components/copilot/chat/AIChatManager.svelte.ts @@ -110,7 +110,7 @@ import type { Selection } from 'monaco-editor' import type AIChatInput from './AIChatInput.svelte' import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core' import { closeInterruptedToolBatch, runChatLoop, truncateToToolPairedPrefix } from './chatLoop' -import { OutputTokenLimitError } from './outputTokenLimit' +import { FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE, OutputTokenLimitError } from './outputTokenLimit' import { sanitizeToolCallArguments } from './toolCallArguments' import { billedTokens, normalizeContextUsage, type ChatTokenUsage } from './tokenUsage' import { logAiUsage } from '$lib/utils/aiUsageReporter' @@ -378,7 +378,7 @@ function isImageRejection(err: unknown, models: (string | undefined)[] = []): bo function getSendRequestErrorMessage(err: unknown, webSearchUnavailable: boolean): string { // The request went through; only its response was cut short. if (err instanceof OutputTokenLimitError) { - return err.message + return get(copilotInfo).freeTier ? FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE : err.message } const errorMessage = err instanceof Error ? err.message : typeof err === 'string' ? err : undefined diff --git a/frontend/src/lib/components/copilot/chat/anthropic.ts b/frontend/src/lib/components/copilot/chat/anthropic.ts index a311d16990..122425e264 100644 --- a/frontend/src/lib/components/copilot/chat/anthropic.ts +++ b/frontend/src/lib/components/copilot/chat/anthropic.ts @@ -141,7 +141,8 @@ export async function getAnthropicCompletion( const { provider, config } = getProviderAndCompletionConfig({ messages, stream: true, - forceModelProvider: options?.forceModelProvider + forceModelProvider: options?.forceModelProvider, + reasoningEffort: options?.reasoningEffort }) const { system, messages: anthropicMessages } = convertOpenAIToAnthropicMessages(messages) let anthropicTools = convertOpenAIToolsToAnthropic(tools) diff --git a/frontend/src/lib/components/copilot/chat/openai-responses.ts b/frontend/src/lib/components/copilot/chat/openai-responses.ts index b216a08107..0e460f1921 100644 --- a/frontend/src/lib/components/copilot/chat/openai-responses.ts +++ b/frontend/src/lib/components/copilot/chat/openai-responses.ts @@ -272,7 +272,8 @@ export async function getOpenAIResponsesCompletion( messages, stream: true, tools, - forceModelProvider: options?.forceModelProvider + forceModelProvider: options?.forceModelProvider, + reasoningEffort: options?.reasoningEffort }) const { instructions, input } = convertMessagesToResponsesInput(messages) const responsesConfig = applyReasoningToConfig( @@ -331,7 +332,8 @@ export async function* getOpenAIResponsesCompletionStream( messages, stream: true, tools, - forceModelProvider: options?.forceModelProvider + forceModelProvider: options?.forceModelProvider, + reasoningEffort: options?.reasoningEffort }) const { instructions, input } = convertMessagesToResponsesInput(messages) // No prompt cache key here: a rejected key has to be retried without it, and this is diff --git a/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts b/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts index b1127a1db7..19b74f0c68 100644 --- a/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts +++ b/frontend/src/lib/components/copilot/chat/outputTokenLimit.ts @@ -10,3 +10,8 @@ export class OutputTokenLimitError extends Error { this.name = 'OutputTokenLimitError' } } + +/** Shown instead on Windmill's free tier, whose server clamps the output below any + * workspace setting: pointing at that setting would send the user to a no-op. */ +export const FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE = + "The response was cut off at the free tier's output token limit. Ask it to continue, or add your own API key in the workspace AI settings for a higher limit." diff --git a/frontend/src/lib/components/copilot/lib.toolCalls.test.ts b/frontend/src/lib/components/copilot/lib.toolCalls.test.ts index b011c1f672..6966e018d1 100644 --- a/frontend/src/lib/components/copilot/lib.toolCalls.test.ts +++ b/frontend/src/lib/components/copilot/lib.toolCalls.test.ts @@ -287,10 +287,42 @@ describe('parseOpenAICompletion tool call arguments', () => { }) describe('output token limit', () => { - it("gives DeepSeek DeepSeek's own thinking-mode budget rather than the 8192 fallback", async () => { + it('follows the model through every host that serves it', async () => { const { getModelMaxTokens } = await import('./lib') expect(getModelMaxTokens('deepseek', 'deepseek-flash')).toBe(131072) - expect(getModelMaxTokens('deepseek', 'deepseek-v4-pro')).toBe(131072) + expect(getModelMaxTokens('azure_foundry', 'DeepSeek-V4-Pro')).toBe(131072) + expect(getModelMaxTokens('openrouter', 'deepseek/deepseek-v4-pro')).toBe(131072) + // gpt-oss reasons whether or not the chat sends an effort. + expect(getModelMaxTokens('groq', 'openai/gpt-oss-120b')).toBe(16000) + expect(getModelMaxTokens('openrouter', 'openai/o3')).toBe(100000) + // A legacy `/thinking` selection is its base model, not the `thinking` row. + expect(getModelMaxTokens('anthropic', 'claude-sonnet-4-5/thinking')).toBe(64000) + }) + + it('gives a model outside the table room to think only when the request asks it to', async () => { + const { getModelMaxTokens } = await import('./lib') + expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3', 'high')).toBe(32768) + expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3', 'none')).toBe(8192) + expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3')).toBe(8192) + // The registry does not know Custom AI models, so an effort says nothing about them. + expect(getModelMaxTokens('customai', 'local-model', 'high')).toBe(8192) + // A self-hosted open-weight model runs at its operator's context length, not the + // vendor's, while a Custom AI proxy in front of a hosted model keeps its row. + expect(getModelMaxTokens('customai', 'deepseek-v4-pro', 'high')).toBe(8192) + expect(getModelMaxTokens('customai', 'gpt-5')).toBe(128000) + }) + + it('sizes the budget from the effort the same request carries', async () => { + const { getCompletion } = await import('./lib') + const create = vi.fn() + + await getCompletion([], new AbortController(), undefined, { + forceModelProvider: { provider: 'mistral', model: 'mistral-medium-latest' }, + openaiClient: { chat: { completions: { create } } } as any, + reasoningEffort: 'high' + }) + + expect(create.mock.calls[0][0]).toMatchObject({ max_tokens: 32768, reasoning_effort: 'high' }) }) it('fails a response cut off at the output token limit instead of ending the turn', async () => { diff --git a/frontend/src/lib/components/copilot/lib.ts b/frontend/src/lib/components/copilot/lib.ts index 5094764807..b55aa06c92 100644 --- a/frontend/src/lib/components/copilot/lib.ts +++ b/frontend/src/lib/components/copilot/lib.ts @@ -15,11 +15,17 @@ import { get, type Writable } from 'svelte/store' import { OpenAPI, ResourceService, type Script } from '../../gen' import { EDIT_CONFIG, FIX_CONFIG, GEN_CONFIG } from './prompts' import { + getKnownModelMaxOutputTokens, requiresMaxCompletionTokens, usesAnthropicMessagesApi, usesOpenRouterPromptCaching } from './modelConfig' -import { applyReasoningToConfig } from './reasoningRegistry' +import { + applyReasoningToConfig, + requestsReasoning, + stripLegacyThinkingSuffix, + type ReasoningEffort +} from './reasoningRegistry' import { formatResourceTypes } from './utils' import { appendPendingToolImages, @@ -310,45 +316,31 @@ export async function fetchAvailableModels( return data?.data.map((m) => m.id) ?? [] } -export function getModelMaxTokens(provider: AIProvider, model: string) { - if (provider === 'deepseek') { - // DeepSeek thinks by default and counts the thinking toward max_tokens, so - // the 8192 fallback cuts long turns off mid-thought. 131072 is DeepSeek's - // own default at the highest effort (65536 at the default one). - return 131072 - } else if (model.includes('gpt-5')) { - return 128000 - } else if ( - (provider === 'azure_openai' || provider === 'openai' || provider === 'azure_foundry') && - model.startsWith('o') - ) { - return 100000 - } else if ( - // Raising this further would also raise the worst case of the - // non-streaming completion path, which the Anthropic SDK refuses once - // the request could run past ~10 minutes. - model.includes('claude-sonnet') || - model.includes('claude-haiku') || - model.includes('claude-fable') || - model.includes('claude-mythos') || - // Opus only from 4.5 on. Opus 4.1 and older cap at 32K and fall through - // to the row below. Dots are normalized because OpenRouter writes - // `anthropic/claude-opus-4.5` where Anthropic writes `claude-opus-4-5`. - /claude-opus-(4-(5|6|7|8)|5)(?!\d)/.test(model.replace(/\./g, '-')) || - model.includes('gemini-2.5') || - model.includes('gemini-3') - ) { - return 64000 - } else if (model.includes('gpt-4.1')) { - return 32768 - } else if (model.includes('claude-opus')) { - return 32000 - } else if (model.includes('gpt-4o') || model.includes('codestral')) { - return 16384 - } else if (model.includes('gpt-4-turbo') || model.includes('gpt-3.5')) { - return 4096 - } - return 8192 +// Thinking counts toward max_tokens, so a model outside the table that is asked to +// reason gets more room than the plain fallback. A host rejects a budget above the +// model's cap, so the reasoning fallback must stay within the cap of every model it +// reaches (true of every reasoning model OpenRouter serves outside the table). Custom +// AI never gets it: the registry does not know its models, and the per-model override +// is its way to a larger budget. +const FALLBACK_MAX_TOKENS = 8192 +const REASONING_FALLBACK_MAX_TOKENS = 32768 + +/** + * The output budget a request sends when the workspace sets no override. `reasoningEffort` + * is the effort the request carries, as `resolveRequestReasoning` resolves it. + */ +export function getModelMaxTokens( + provider: AIProvider, + model: string, + reasoningEffort?: ReasoningEffort +): number { + const bareModel = stripLegacyThinkingSuffix(model) + return ( + getKnownModelMaxOutputTokens(provider, bareModel) ?? + (requestsReasoning(provider, bareModel, reasoningEffort) + ? REASONING_FALLBACK_MAX_TOKENS + : FALLBACK_MAX_TOKENS) + ) } // Resolves the completion token cap for a model: the workspace's per-model @@ -356,8 +348,16 @@ export function getModelMaxTokens(provider: AIProvider, model: string) { // Anthropic request paths so both honor the same limit. `cap` bounds the result // (used by short metadata completions, see METADATA_MAX_TOKENS) — a hard ceiling // that wins over both the workspace override and the default. -function resolveMaxTokens(modelProvider: AIProviderModel, cap?: number): number { - const defaultMaxTokens = getModelMaxTokens(modelProvider.provider, modelProvider.model) +function resolveMaxTokens( + modelProvider: AIProviderModel, + cap?: number, + reasoningEffort?: ReasoningEffort +): number { + const defaultMaxTokens = getModelMaxTokens( + modelProvider.provider, + modelProvider.model, + reasoningEffort + ) const modelKey = `${modelProvider.provider}:${modelProvider.model}` let customMaxTokensStore: Record | undefined try { @@ -379,9 +379,10 @@ export const METADATA_MAX_TOKENS = 4096 function getModelSpecificConfig( modelProvider: AIProviderModel, tools?: OpenAI.Chat.Completions.ChatCompletionTool[], - maxTokensCap?: number + maxTokensCap?: number, + reasoningEffort?: ReasoningEffort ) { - const maxTokens = resolveMaxTokens(modelProvider, maxTokensCap) + const maxTokens = resolveMaxTokens(modelProvider, maxTokensCap, reasoningEffort) if ( (modelProvider.provider === 'openai' || modelProvider.provider === 'azure_openai' || @@ -903,7 +904,8 @@ export function getProviderAndCompletionConfig({ tools, forceModelProvider, maxTokensCap, - promptCaching + promptCaching, + reasoningEffort }: { messages: ChatCompletionMessageParam[] stream: K @@ -913,6 +915,9 @@ export function getProviderAndCompletionConfig({ // Opt-in: a cache write costs more than an uncached read, so it only pays off where // the same prefix is sent again. True for the chat loop, false for one-shot calls. promptCaching?: boolean + // The effort the caller will add with `applyReasoningToConfig`. Only sizes the + // output budget here: a request that reasons needs room for the thinking. + reasoningEffort?: ReasoningEffort }): { provider: AIProvider config: K extends true @@ -929,7 +934,7 @@ export function getProviderAndCompletionConfig({ provider: modelProvider.provider, config: { ...providerConfig, - ...getModelSpecificConfig(modelProvider, tools, maxTokensCap), + ...getModelSpecificConfig(modelProvider, tools, maxTokensCap, reasoningEffort), messages: processedMessages, stream } as any @@ -1135,7 +1140,8 @@ export async function getCompletion( stream: true, tools, forceModelProvider: options?.forceModelProvider, - promptCaching: options?.promptCaching + promptCaching: options?.promptCaching, + reasoningEffort: options?.reasoningEffort }) // Use Responses API for OpenAI and Azure OpenAI diff --git a/frontend/src/lib/components/copilot/modelConfig.ts b/frontend/src/lib/components/copilot/modelConfig.ts index be49907ed4..77d0cf942a 100644 --- a/frontend/src/lib/components/copilot/modelConfig.ts +++ b/frontend/src/lib/components/copilot/modelConfig.ts @@ -117,6 +117,58 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [ ['codestral', 32_000] ] +// Output token budgets, matched like the context windows above so one row covers +// every host of a model: native API, Azure AI Foundry, OpenRouter, Together, Bedrock. +// A host rejects any request above its own limit, so each value is one every host of +// the model accepts, not the vendor's maximum. Models not listed fall back in +// `getModelMaxTokens`. +const MODEL_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [ + // Anthropic — the Anthropic SDK refuses a non-streaming request whose max_tokens + // implies more than ~10 minutes, so these stay below the models' own maximum. + // Opus caps at 32K before 4.5. + ['claude-opus-4.5', 64_000], + ['claude-opus-4.6', 64_000], + ['claude-opus-4.7', 64_000], + ['claude-opus-4.8', 64_000], + ['claude-opus-5', 64_000], + ['claude-opus', 32_000], + ['claude-sonnet', 64_000], + ['claude-haiku', 64_000], + ['claude-fable', 64_000], + ['claude-mythos', 64_000], + // OpenAI + ['gpt-5', 128_000], + ['gpt-4.1', 32_768], + ['gpt-4o', 16_384], + ['gpt-4-turbo', 4_096], + ['gpt-3.5', 4_096], + ['o1', 100_000], + ['o3', 100_000], + ['o4-mini', 100_000], + // Google + ['gemini-3', 64_000], + ['gemini-2.5', 64_000] +] + +// Open-weight models, which a self-hosted server (Custom AI) caps at whatever context +// length its operator set, often below these: vLLM rejects a request whose prompt plus +// max_tokens exceeds it. So these rows only apply to the providers that host them. +const OPEN_WEIGHT_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [ + // gpt-oss reasons on every request. Bedrock documents 16K, where Groq takes 65536 + ['gpt-oss', 16_000], + // DeepSeek — thinks by default and counts the thinking toward max_tokens. 131072 + // is DeepSeek's own default at the highest effort. R1 is capped by its OpenRouter + // host. + ['deepseek-v4', 131_072], + ['deepseek-flash', 131_072], + ['deepseek-r1', 16_000], + // Mistral + ['codestral', 16_384], + // Models named for thinking reason on every request, whether or not the chat sends + // an effort. Every one OpenRouter lists takes at least this much. + ['thinking', 32_768] +] + // Version separators differ by route to the same model: Anthropic writes // `claude-opus-4-8`, OpenRouter writes `anthropic/claude-opus-4.8`. Collapsing // dots to dashes on both sides keeps one table entry covering every route — @@ -197,6 +249,19 @@ export function getKnownModelContextWindow(model: string): number | undefined { return matchModel(MODEL_CONTEXT_WINDOW_MATCHERS, model) } +const MODEL_MAX_OUTPUT_TOKEN_MATCHERS = buildModelMatchers(MODEL_MAX_OUTPUT_TOKENS) +const OPEN_WEIGHT_MAX_OUTPUT_TOKEN_MATCHERS = buildModelMatchers(OPEN_WEIGHT_MAX_OUTPUT_TOKENS) + +export function getKnownModelMaxOutputTokens( + provider: AIProvider, + model: string +): number | undefined { + return ( + matchModel(MODEL_MAX_OUTPUT_TOKEN_MATCHERS, model) ?? + (provider === 'customai' ? undefined : matchModel(OPEN_WEIGHT_MAX_OUTPUT_TOKEN_MATCHERS, model)) + ) +} + /** Trim/compaction logic needs a number; assume a conservative window when unknown. */ export const ASSUMED_CONTEXT_WINDOW = 128000 diff --git a/frontend/src/lib/components/copilot/reasoningRegistry.ts b/frontend/src/lib/components/copilot/reasoningRegistry.ts index 242f6cb329..9915e9c42f 100644 --- a/frontend/src/lib/components/copilot/reasoningRegistry.ts +++ b/frontend/src/lib/components/copilot/reasoningRegistry.ts @@ -374,7 +374,8 @@ const ANTHROPIC_ALWAYS_THINKING = /fable|mythos/ /** * Disable token to forward when the user explicitly turns reasoning off on a * model that reasons *by default* — omitting the field would silently keep - * the default-on behavior. Undefined means omission is the correct off. + * the default-on behavior. Undefined means omission is the correct off. Every + * token is `'none'`: `requestsReasoning` reads that value as off. */ export function explicitOffToken(provider: AIProvider, model: string): ReasoningEffort | undefined { switch (reasoningProviderFamily(provider, model)) { @@ -434,6 +435,20 @@ export function resolveRequestReasoning( return undefined } +/** + * Whether a request carrying `effort`, as `resolveRequestReasoning` resolves it, + * has the model reason. An effort sent to a model the registry does not mark as + * reasoning-capable answers no: an explicit choice is always sent, but that does + * not make the model think. + */ +export function requestsReasoning( + provider: AIProvider, + model: string, + effort: ReasoningEffort | undefined +): boolean { + return !!effort && effort !== 'none' && supportsReasoning(provider, model) +} + export type ReasoningApiKind = 'anthropic' | 'responses' | 'completions' | 'deepseek' | 'mistral' /** diff --git a/frontend/src/lib/components/workspaceSettings/AISettings.svelte b/frontend/src/lib/components/workspaceSettings/AISettings.svelte index 3957f47b74..05159e3057 100644 --- a/frontend/src/lib/components/workspaceSettings/AISettings.svelte +++ b/frontend/src/lib/components/workspaceSettings/AISettings.svelte @@ -21,6 +21,7 @@ getKnownModelContextWindow, getModelContextWindowFromTable } from '../copilot/modelConfig' + import { resolveRequestReasoning } from '../copilot/reasoningRegistry' import { supportsAutocomplete } from '../copilot/utils' import TestAiKey from '../copilot/TestAIKey.svelte' import Label from '../Label.svelte' @@ -650,7 +651,8 @@ min={1} max={2_000_000} getDefault={(provider, model) => ({ - tokens: getModelMaxTokens(provider, model), + // What a chat sends at the default effort; the budget depends on it. + tokens: getModelMaxTokens(provider, model, resolveRequestReasoning({ provider, model })), assumed: false })} />