fix: size the AI chat output budget by model, not provider (#11286)

* fix: size the AI chat output budget by model, not provider

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* fix: keep self-hosted open-weight models on the fallback output budget

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* chore: point ee-repo-ref at the free tier output clamp change

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* fix: treat codestral as open-weight for the output budget

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* chore: update ee-repo-ref to 559a9ff78ea03439553cf2b412e73765762e9013

This commit updates the EE repository reference after PR #821 was merged in windmill-ee-private.

Previous ee-repo-ref: 953f7c1cc9a70b740d5d1b129cb1401950854b02

New ee-repo-ref: 559a9ff78ea03439553cf2b412e73765762e9013

Automated by sync-ee-ref workflow.

---------

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
Co-authored-by: windmill-internal-app[bot] <windmill-internal-app[bot]@users.noreply.github.com>
This commit is contained in:
Ruben Fiszel
2026-09-22 09:32:11 +00:00
committed by GitHub
co-authored by Claude Opus 5 windmill-internal-app[bot]
parent efa7a0a70a
commit 7d0f2ce3bf
10 changed files with 185 additions and 57 deletions
+1 -1
View File
@@ -1 +1 @@
5370ae3a95a3dc10171f72da74286a654996eee6
559a9ff78ea03439553cf2b412e73765762e9013
@@ -110,7 +110,7 @@ import type { Selection } from 'monaco-editor'
import type AIChatInput from './AIChatInput.svelte'
import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core'
import { closeInterruptedToolBatch, runChatLoop, truncateToToolPairedPrefix } from './chatLoop'
import { OutputTokenLimitError } from './outputTokenLimit'
import { FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE, OutputTokenLimitError } from './outputTokenLimit'
import { sanitizeToolCallArguments } from './toolCallArguments'
import { billedTokens, normalizeContextUsage, type ChatTokenUsage } from './tokenUsage'
import { logAiUsage } from '$lib/utils/aiUsageReporter'
@@ -378,7 +378,7 @@ function isImageRejection(err: unknown, models: (string | undefined)[] = []): bo
function getSendRequestErrorMessage(err: unknown, webSearchUnavailable: boolean): string {
// The request went through; only its response was cut short.
if (err instanceof OutputTokenLimitError) {
return err.message
return get(copilotInfo).freeTier ? FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE : err.message
}
const errorMessage =
err instanceof Error ? err.message : typeof err === 'string' ? err : undefined
@@ -141,7 +141,8 @@ export async function getAnthropicCompletion(
const { provider, config } = getProviderAndCompletionConfig({
messages,
stream: true,
forceModelProvider: options?.forceModelProvider
forceModelProvider: options?.forceModelProvider,
reasoningEffort: options?.reasoningEffort
})
const { system, messages: anthropicMessages } = convertOpenAIToAnthropicMessages(messages)
let anthropicTools = convertOpenAIToolsToAnthropic(tools)
@@ -272,7 +272,8 @@ export async function getOpenAIResponsesCompletion(
messages,
stream: true,
tools,
forceModelProvider: options?.forceModelProvider
forceModelProvider: options?.forceModelProvider,
reasoningEffort: options?.reasoningEffort
})
const { instructions, input } = convertMessagesToResponsesInput(messages)
const responsesConfig = applyReasoningToConfig(
@@ -331,7 +332,8 @@ export async function* getOpenAIResponsesCompletionStream(
messages,
stream: true,
tools,
forceModelProvider: options?.forceModelProvider
forceModelProvider: options?.forceModelProvider,
reasoningEffort: options?.reasoningEffort
})
const { instructions, input } = convertMessagesToResponsesInput(messages)
// No prompt cache key here: a rejected key has to be retried without it, and this is
@@ -10,3 +10,8 @@ export class OutputTokenLimitError extends Error {
this.name = 'OutputTokenLimitError'
}
}
/** Shown instead on Windmill's free tier, whose server clamps the output below any
* workspace setting: pointing at that setting would send the user to a no-op. */
export const FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE =
"The response was cut off at the free tier's output token limit. Ask it to continue, or add your own API key in the workspace AI settings for a higher limit."
@@ -287,10 +287,42 @@ describe('parseOpenAICompletion tool call arguments', () => {
})
describe('output token limit', () => {
it("gives DeepSeek DeepSeek's own thinking-mode budget rather than the 8192 fallback", async () => {
it('follows the model through every host that serves it', async () => {
const { getModelMaxTokens } = await import('./lib')
expect(getModelMaxTokens('deepseek', 'deepseek-flash')).toBe(131072)
expect(getModelMaxTokens('deepseek', 'deepseek-v4-pro')).toBe(131072)
expect(getModelMaxTokens('azure_foundry', 'DeepSeek-V4-Pro')).toBe(131072)
expect(getModelMaxTokens('openrouter', 'deepseek/deepseek-v4-pro')).toBe(131072)
// gpt-oss reasons whether or not the chat sends an effort.
expect(getModelMaxTokens('groq', 'openai/gpt-oss-120b')).toBe(16000)
expect(getModelMaxTokens('openrouter', 'openai/o3')).toBe(100000)
// A legacy `/thinking` selection is its base model, not the `thinking` row.
expect(getModelMaxTokens('anthropic', 'claude-sonnet-4-5/thinking')).toBe(64000)
})
it('gives a model outside the table room to think only when the request asks it to', async () => {
const { getModelMaxTokens } = await import('./lib')
expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3', 'high')).toBe(32768)
expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3', 'none')).toBe(8192)
expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3')).toBe(8192)
// The registry does not know Custom AI models, so an effort says nothing about them.
expect(getModelMaxTokens('customai', 'local-model', 'high')).toBe(8192)
// A self-hosted open-weight model runs at its operator's context length, not the
// vendor's, while a Custom AI proxy in front of a hosted model keeps its row.
expect(getModelMaxTokens('customai', 'deepseek-v4-pro', 'high')).toBe(8192)
expect(getModelMaxTokens('customai', 'gpt-5')).toBe(128000)
})
it('sizes the budget from the effort the same request carries', async () => {
const { getCompletion } = await import('./lib')
const create = vi.fn()
await getCompletion([], new AbortController(), undefined, {
forceModelProvider: { provider: 'mistral', model: 'mistral-medium-latest' },
openaiClient: { chat: { completions: { create } } } as any,
reasoningEffort: 'high'
})
expect(create.mock.calls[0][0]).toMatchObject({ max_tokens: 32768, reasoning_effort: 'high' })
})
it('fails a response cut off at the output token limit instead of ending the turn', async () => {
+53 -47
View File
@@ -15,11 +15,17 @@ import { get, type Writable } from 'svelte/store'
import { OpenAPI, ResourceService, type Script } from '../../gen'
import { EDIT_CONFIG, FIX_CONFIG, GEN_CONFIG } from './prompts'
import {
getKnownModelMaxOutputTokens,
requiresMaxCompletionTokens,
usesAnthropicMessagesApi,
usesOpenRouterPromptCaching
} from './modelConfig'
import { applyReasoningToConfig } from './reasoningRegistry'
import {
applyReasoningToConfig,
requestsReasoning,
stripLegacyThinkingSuffix,
type ReasoningEffort
} from './reasoningRegistry'
import { formatResourceTypes } from './utils'
import {
appendPendingToolImages,
@@ -310,45 +316,31 @@ export async function fetchAvailableModels(
return data?.data.map((m) => m.id) ?? []
}
export function getModelMaxTokens(provider: AIProvider, model: string) {
if (provider === 'deepseek') {
// DeepSeek thinks by default and counts the thinking toward max_tokens, so
// the 8192 fallback cuts long turns off mid-thought. 131072 is DeepSeek's
// own default at the highest effort (65536 at the default one).
return 131072
} else if (model.includes('gpt-5')) {
return 128000
} else if (
(provider === 'azure_openai' || provider === 'openai' || provider === 'azure_foundry') &&
model.startsWith('o')
) {
return 100000
} else if (
// Raising this further would also raise the worst case of the
// non-streaming completion path, which the Anthropic SDK refuses once
// the request could run past ~10 minutes.
model.includes('claude-sonnet') ||
model.includes('claude-haiku') ||
model.includes('claude-fable') ||
model.includes('claude-mythos') ||
// Opus only from 4.5 on. Opus 4.1 and older cap at 32K and fall through
// to the row below. Dots are normalized because OpenRouter writes
// `anthropic/claude-opus-4.5` where Anthropic writes `claude-opus-4-5`.
/claude-opus-(4-(5|6|7|8)|5)(?!\d)/.test(model.replace(/\./g, '-')) ||
model.includes('gemini-2.5') ||
model.includes('gemini-3')
) {
return 64000
} else if (model.includes('gpt-4.1')) {
return 32768
} else if (model.includes('claude-opus')) {
return 32000
} else if (model.includes('gpt-4o') || model.includes('codestral')) {
return 16384
} else if (model.includes('gpt-4-turbo') || model.includes('gpt-3.5')) {
return 4096
}
return 8192
// Thinking counts toward max_tokens, so a model outside the table that is asked to
// reason gets more room than the plain fallback. A host rejects a budget above the
// model's cap, so the reasoning fallback must stay within the cap of every model it
// reaches (true of every reasoning model OpenRouter serves outside the table). Custom
// AI never gets it: the registry does not know its models, and the per-model override
// is its way to a larger budget.
const FALLBACK_MAX_TOKENS = 8192
const REASONING_FALLBACK_MAX_TOKENS = 32768
/**
* The output budget a request sends when the workspace sets no override. `reasoningEffort`
* is the effort the request carries, as `resolveRequestReasoning` resolves it.
*/
export function getModelMaxTokens(
provider: AIProvider,
model: string,
reasoningEffort?: ReasoningEffort
): number {
const bareModel = stripLegacyThinkingSuffix(model)
return (
getKnownModelMaxOutputTokens(provider, bareModel) ??
(requestsReasoning(provider, bareModel, reasoningEffort)
? REASONING_FALLBACK_MAX_TOKENS
: FALLBACK_MAX_TOKENS)
)
}
// Resolves the completion token cap for a model: the workspace's per-model
@@ -356,8 +348,16 @@ export function getModelMaxTokens(provider: AIProvider, model: string) {
// Anthropic request paths so both honor the same limit. `cap` bounds the result
// (used by short metadata completions, see METADATA_MAX_TOKENS) — a hard ceiling
// that wins over both the workspace override and the default.
function resolveMaxTokens(modelProvider: AIProviderModel, cap?: number): number {
const defaultMaxTokens = getModelMaxTokens(modelProvider.provider, modelProvider.model)
function resolveMaxTokens(
modelProvider: AIProviderModel,
cap?: number,
reasoningEffort?: ReasoningEffort
): number {
const defaultMaxTokens = getModelMaxTokens(
modelProvider.provider,
modelProvider.model,
reasoningEffort
)
const modelKey = `${modelProvider.provider}:${modelProvider.model}`
let customMaxTokensStore: Record<string, number> | undefined
try {
@@ -379,9 +379,10 @@ export const METADATA_MAX_TOKENS = 4096
function getModelSpecificConfig(
modelProvider: AIProviderModel,
tools?: OpenAI.Chat.Completions.ChatCompletionTool[],
maxTokensCap?: number
maxTokensCap?: number,
reasoningEffort?: ReasoningEffort
) {
const maxTokens = resolveMaxTokens(modelProvider, maxTokensCap)
const maxTokens = resolveMaxTokens(modelProvider, maxTokensCap, reasoningEffort)
if (
(modelProvider.provider === 'openai' ||
modelProvider.provider === 'azure_openai' ||
@@ -903,7 +904,8 @@ export function getProviderAndCompletionConfig<K extends boolean>({
tools,
forceModelProvider,
maxTokensCap,
promptCaching
promptCaching,
reasoningEffort
}: {
messages: ChatCompletionMessageParam[]
stream: K
@@ -913,6 +915,9 @@ export function getProviderAndCompletionConfig<K extends boolean>({
// Opt-in: a cache write costs more than an uncached read, so it only pays off where
// the same prefix is sent again. True for the chat loop, false for one-shot calls.
promptCaching?: boolean
// The effort the caller will add with `applyReasoningToConfig`. Only sizes the
// output budget here: a request that reasons needs room for the thinking.
reasoningEffort?: ReasoningEffort
}): {
provider: AIProvider
config: K extends true
@@ -929,7 +934,7 @@ export function getProviderAndCompletionConfig<K extends boolean>({
provider: modelProvider.provider,
config: {
...providerConfig,
...getModelSpecificConfig(modelProvider, tools, maxTokensCap),
...getModelSpecificConfig(modelProvider, tools, maxTokensCap, reasoningEffort),
messages: processedMessages,
stream
} as any
@@ -1135,7 +1140,8 @@ export async function getCompletion(
stream: true,
tools,
forceModelProvider: options?.forceModelProvider,
promptCaching: options?.promptCaching
promptCaching: options?.promptCaching,
reasoningEffort: options?.reasoningEffort
})
// Use Responses API for OpenAI and Azure OpenAI
@@ -117,6 +117,58 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
['codestral', 32_000]
]
// Output token budgets, matched like the context windows above so one row covers
// every host of a model: native API, Azure AI Foundry, OpenRouter, Together, Bedrock.
// A host rejects any request above its own limit, so each value is one every host of
// the model accepts, not the vendor's maximum. Models not listed fall back in
// `getModelMaxTokens`.
const MODEL_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [
// Anthropic — the Anthropic SDK refuses a non-streaming request whose max_tokens
// implies more than ~10 minutes, so these stay below the models' own maximum.
// Opus caps at 32K before 4.5.
['claude-opus-4.5', 64_000],
['claude-opus-4.6', 64_000],
['claude-opus-4.7', 64_000],
['claude-opus-4.8', 64_000],
['claude-opus-5', 64_000],
['claude-opus', 32_000],
['claude-sonnet', 64_000],
['claude-haiku', 64_000],
['claude-fable', 64_000],
['claude-mythos', 64_000],
// OpenAI
['gpt-5', 128_000],
['gpt-4.1', 32_768],
['gpt-4o', 16_384],
['gpt-4-turbo', 4_096],
['gpt-3.5', 4_096],
['o1', 100_000],
['o3', 100_000],
['o4-mini', 100_000],
// Google
['gemini-3', 64_000],
['gemini-2.5', 64_000]
]
// Open-weight models, which a self-hosted server (Custom AI) caps at whatever context
// length its operator set, often below these: vLLM rejects a request whose prompt plus
// max_tokens exceeds it. So these rows only apply to the providers that host them.
const OPEN_WEIGHT_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [
// gpt-oss reasons on every request. Bedrock documents 16K, where Groq takes 65536
['gpt-oss', 16_000],
// DeepSeek — thinks by default and counts the thinking toward max_tokens. 131072
// is DeepSeek's own default at the highest effort. R1 is capped by its OpenRouter
// host.
['deepseek-v4', 131_072],
['deepseek-flash', 131_072],
['deepseek-r1', 16_000],
// Mistral
['codestral', 16_384],
// Models named for thinking reason on every request, whether or not the chat sends
// an effort. Every one OpenRouter lists takes at least this much.
['thinking', 32_768]
]
// Version separators differ by route to the same model: Anthropic writes
// `claude-opus-4-8`, OpenRouter writes `anthropic/claude-opus-4.8`. Collapsing
// dots to dashes on both sides keeps one table entry covering every route —
@@ -197,6 +249,19 @@ export function getKnownModelContextWindow(model: string): number | undefined {
return matchModel(MODEL_CONTEXT_WINDOW_MATCHERS, model)
}
const MODEL_MAX_OUTPUT_TOKEN_MATCHERS = buildModelMatchers(MODEL_MAX_OUTPUT_TOKENS)
const OPEN_WEIGHT_MAX_OUTPUT_TOKEN_MATCHERS = buildModelMatchers(OPEN_WEIGHT_MAX_OUTPUT_TOKENS)
export function getKnownModelMaxOutputTokens(
provider: AIProvider,
model: string
): number | undefined {
return (
matchModel(MODEL_MAX_OUTPUT_TOKEN_MATCHERS, model) ??
(provider === 'customai' ? undefined : matchModel(OPEN_WEIGHT_MAX_OUTPUT_TOKEN_MATCHERS, model))
)
}
/** Trim/compaction logic needs a number; assume a conservative window when unknown. */
export const ASSUMED_CONTEXT_WINDOW = 128000
@@ -374,7 +374,8 @@ const ANTHROPIC_ALWAYS_THINKING = /fable|mythos/
/**
* Disable token to forward when the user explicitly turns reasoning off on a
* model that reasons *by default* — omitting the field would silently keep
* the default-on behavior. Undefined means omission is the correct off.
* the default-on behavior. Undefined means omission is the correct off. Every
* token is `'none'`: `requestsReasoning` reads that value as off.
*/
export function explicitOffToken(provider: AIProvider, model: string): ReasoningEffort | undefined {
switch (reasoningProviderFamily(provider, model)) {
@@ -434,6 +435,20 @@ export function resolveRequestReasoning(
return undefined
}
/**
* Whether a request carrying `effort`, as `resolveRequestReasoning` resolves it,
* has the model reason. An effort sent to a model the registry does not mark as
* reasoning-capable answers no: an explicit choice is always sent, but that does
* not make the model think.
*/
export function requestsReasoning(
provider: AIProvider,
model: string,
effort: ReasoningEffort | undefined
): boolean {
return !!effort && effort !== 'none' && supportsReasoning(provider, model)
}
export type ReasoningApiKind = 'anthropic' | 'responses' | 'completions' | 'deepseek' | 'mistral'
/**
@@ -21,6 +21,7 @@
getKnownModelContextWindow,
getModelContextWindowFromTable
} from '../copilot/modelConfig'
import { resolveRequestReasoning } from '../copilot/reasoningRegistry'
import { supportsAutocomplete } from '../copilot/utils'
import TestAiKey from '../copilot/TestAIKey.svelte'
import Label from '../Label.svelte'
@@ -650,7 +651,8 @@
min={1}
max={2_000_000}
getDefault={(provider, model) => ({
tokens: getModelMaxTokens(provider, model),
// What a chat sends at the default effort; the budget depends on it.
tokens: getModelMaxTokens(provider, model, resolveRequestReasoning({ provider, model })),
assumed: false
})}
/>