mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-10-03 08:02:19 +00:00
fix: size the AI chat output budget by model, not provider (#11286)
* fix: size the AI chat output budget by model, not provider Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * fix: keep self-hosted open-weight models on the fallback output budget Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * chore: point ee-repo-ref at the free tier output clamp change Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * fix: treat codestral as open-weight for the output budget Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * chore: update ee-repo-ref to 559a9ff78ea03439553cf2b412e73765762e9013 This commit updates the EE repository reference after PR #821 was merged in windmill-ee-private. Previous ee-repo-ref: 953f7c1cc9a70b740d5d1b129cb1401950854b02 New ee-repo-ref: 559a9ff78ea03439553cf2b412e73765762e9013 Automated by sync-ee-ref workflow. --------- Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com> Co-authored-by: windmill-internal-app[bot] <windmill-internal-app[bot]@users.noreply.github.com>
This commit is contained in:
co-authored by
Claude Opus 5
windmill-internal-app[bot]
parent
efa7a0a70a
commit
7d0f2ce3bf
@@ -1 +1 @@
|
||||
5370ae3a95a3dc10171f72da74286a654996eee6
|
||||
559a9ff78ea03439553cf2b412e73765762e9013
|
||||
|
||||
@@ -110,7 +110,7 @@ import type { Selection } from 'monaco-editor'
|
||||
import type AIChatInput from './AIChatInput.svelte'
|
||||
import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core'
|
||||
import { closeInterruptedToolBatch, runChatLoop, truncateToToolPairedPrefix } from './chatLoop'
|
||||
import { OutputTokenLimitError } from './outputTokenLimit'
|
||||
import { FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE, OutputTokenLimitError } from './outputTokenLimit'
|
||||
import { sanitizeToolCallArguments } from './toolCallArguments'
|
||||
import { billedTokens, normalizeContextUsage, type ChatTokenUsage } from './tokenUsage'
|
||||
import { logAiUsage } from '$lib/utils/aiUsageReporter'
|
||||
@@ -378,7 +378,7 @@ function isImageRejection(err: unknown, models: (string | undefined)[] = []): bo
|
||||
function getSendRequestErrorMessage(err: unknown, webSearchUnavailable: boolean): string {
|
||||
// The request went through; only its response was cut short.
|
||||
if (err instanceof OutputTokenLimitError) {
|
||||
return err.message
|
||||
return get(copilotInfo).freeTier ? FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE : err.message
|
||||
}
|
||||
const errorMessage =
|
||||
err instanceof Error ? err.message : typeof err === 'string' ? err : undefined
|
||||
|
||||
@@ -141,7 +141,8 @@ export async function getAnthropicCompletion(
|
||||
const { provider, config } = getProviderAndCompletionConfig({
|
||||
messages,
|
||||
stream: true,
|
||||
forceModelProvider: options?.forceModelProvider
|
||||
forceModelProvider: options?.forceModelProvider,
|
||||
reasoningEffort: options?.reasoningEffort
|
||||
})
|
||||
const { system, messages: anthropicMessages } = convertOpenAIToAnthropicMessages(messages)
|
||||
let anthropicTools = convertOpenAIToolsToAnthropic(tools)
|
||||
|
||||
@@ -272,7 +272,8 @@ export async function getOpenAIResponsesCompletion(
|
||||
messages,
|
||||
stream: true,
|
||||
tools,
|
||||
forceModelProvider: options?.forceModelProvider
|
||||
forceModelProvider: options?.forceModelProvider,
|
||||
reasoningEffort: options?.reasoningEffort
|
||||
})
|
||||
const { instructions, input } = convertMessagesToResponsesInput(messages)
|
||||
const responsesConfig = applyReasoningToConfig(
|
||||
@@ -331,7 +332,8 @@ export async function* getOpenAIResponsesCompletionStream(
|
||||
messages,
|
||||
stream: true,
|
||||
tools,
|
||||
forceModelProvider: options?.forceModelProvider
|
||||
forceModelProvider: options?.forceModelProvider,
|
||||
reasoningEffort: options?.reasoningEffort
|
||||
})
|
||||
const { instructions, input } = convertMessagesToResponsesInput(messages)
|
||||
// No prompt cache key here: a rejected key has to be retried without it, and this is
|
||||
|
||||
@@ -10,3 +10,8 @@ export class OutputTokenLimitError extends Error {
|
||||
this.name = 'OutputTokenLimitError'
|
||||
}
|
||||
}
|
||||
|
||||
/** Shown instead on Windmill's free tier, whose server clamps the output below any
|
||||
* workspace setting: pointing at that setting would send the user to a no-op. */
|
||||
export const FREE_TIER_OUTPUT_TOKEN_LIMIT_MESSAGE =
|
||||
"The response was cut off at the free tier's output token limit. Ask it to continue, or add your own API key in the workspace AI settings for a higher limit."
|
||||
|
||||
@@ -287,10 +287,42 @@ describe('parseOpenAICompletion tool call arguments', () => {
|
||||
})
|
||||
|
||||
describe('output token limit', () => {
|
||||
it("gives DeepSeek DeepSeek's own thinking-mode budget rather than the 8192 fallback", async () => {
|
||||
it('follows the model through every host that serves it', async () => {
|
||||
const { getModelMaxTokens } = await import('./lib')
|
||||
expect(getModelMaxTokens('deepseek', 'deepseek-flash')).toBe(131072)
|
||||
expect(getModelMaxTokens('deepseek', 'deepseek-v4-pro')).toBe(131072)
|
||||
expect(getModelMaxTokens('azure_foundry', 'DeepSeek-V4-Pro')).toBe(131072)
|
||||
expect(getModelMaxTokens('openrouter', 'deepseek/deepseek-v4-pro')).toBe(131072)
|
||||
// gpt-oss reasons whether or not the chat sends an effort.
|
||||
expect(getModelMaxTokens('groq', 'openai/gpt-oss-120b')).toBe(16000)
|
||||
expect(getModelMaxTokens('openrouter', 'openai/o3')).toBe(100000)
|
||||
// A legacy `/thinking` selection is its base model, not the `thinking` row.
|
||||
expect(getModelMaxTokens('anthropic', 'claude-sonnet-4-5/thinking')).toBe(64000)
|
||||
})
|
||||
|
||||
it('gives a model outside the table room to think only when the request asks it to', async () => {
|
||||
const { getModelMaxTokens } = await import('./lib')
|
||||
expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3', 'high')).toBe(32768)
|
||||
expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3', 'none')).toBe(8192)
|
||||
expect(getModelMaxTokens('openrouter', 'x-ai/grok-4.3')).toBe(8192)
|
||||
// The registry does not know Custom AI models, so an effort says nothing about them.
|
||||
expect(getModelMaxTokens('customai', 'local-model', 'high')).toBe(8192)
|
||||
// A self-hosted open-weight model runs at its operator's context length, not the
|
||||
// vendor's, while a Custom AI proxy in front of a hosted model keeps its row.
|
||||
expect(getModelMaxTokens('customai', 'deepseek-v4-pro', 'high')).toBe(8192)
|
||||
expect(getModelMaxTokens('customai', 'gpt-5')).toBe(128000)
|
||||
})
|
||||
|
||||
it('sizes the budget from the effort the same request carries', async () => {
|
||||
const { getCompletion } = await import('./lib')
|
||||
const create = vi.fn()
|
||||
|
||||
await getCompletion([], new AbortController(), undefined, {
|
||||
forceModelProvider: { provider: 'mistral', model: 'mistral-medium-latest' },
|
||||
openaiClient: { chat: { completions: { create } } } as any,
|
||||
reasoningEffort: 'high'
|
||||
})
|
||||
|
||||
expect(create.mock.calls[0][0]).toMatchObject({ max_tokens: 32768, reasoning_effort: 'high' })
|
||||
})
|
||||
|
||||
it('fails a response cut off at the output token limit instead of ending the turn', async () => {
|
||||
|
||||
@@ -15,11 +15,17 @@ import { get, type Writable } from 'svelte/store'
|
||||
import { OpenAPI, ResourceService, type Script } from '../../gen'
|
||||
import { EDIT_CONFIG, FIX_CONFIG, GEN_CONFIG } from './prompts'
|
||||
import {
|
||||
getKnownModelMaxOutputTokens,
|
||||
requiresMaxCompletionTokens,
|
||||
usesAnthropicMessagesApi,
|
||||
usesOpenRouterPromptCaching
|
||||
} from './modelConfig'
|
||||
import { applyReasoningToConfig } from './reasoningRegistry'
|
||||
import {
|
||||
applyReasoningToConfig,
|
||||
requestsReasoning,
|
||||
stripLegacyThinkingSuffix,
|
||||
type ReasoningEffort
|
||||
} from './reasoningRegistry'
|
||||
import { formatResourceTypes } from './utils'
|
||||
import {
|
||||
appendPendingToolImages,
|
||||
@@ -310,45 +316,31 @@ export async function fetchAvailableModels(
|
||||
return data?.data.map((m) => m.id) ?? []
|
||||
}
|
||||
|
||||
export function getModelMaxTokens(provider: AIProvider, model: string) {
|
||||
if (provider === 'deepseek') {
|
||||
// DeepSeek thinks by default and counts the thinking toward max_tokens, so
|
||||
// the 8192 fallback cuts long turns off mid-thought. 131072 is DeepSeek's
|
||||
// own default at the highest effort (65536 at the default one).
|
||||
return 131072
|
||||
} else if (model.includes('gpt-5')) {
|
||||
return 128000
|
||||
} else if (
|
||||
(provider === 'azure_openai' || provider === 'openai' || provider === 'azure_foundry') &&
|
||||
model.startsWith('o')
|
||||
) {
|
||||
return 100000
|
||||
} else if (
|
||||
// Raising this further would also raise the worst case of the
|
||||
// non-streaming completion path, which the Anthropic SDK refuses once
|
||||
// the request could run past ~10 minutes.
|
||||
model.includes('claude-sonnet') ||
|
||||
model.includes('claude-haiku') ||
|
||||
model.includes('claude-fable') ||
|
||||
model.includes('claude-mythos') ||
|
||||
// Opus only from 4.5 on. Opus 4.1 and older cap at 32K and fall through
|
||||
// to the row below. Dots are normalized because OpenRouter writes
|
||||
// `anthropic/claude-opus-4.5` where Anthropic writes `claude-opus-4-5`.
|
||||
/claude-opus-(4-(5|6|7|8)|5)(?!\d)/.test(model.replace(/\./g, '-')) ||
|
||||
model.includes('gemini-2.5') ||
|
||||
model.includes('gemini-3')
|
||||
) {
|
||||
return 64000
|
||||
} else if (model.includes('gpt-4.1')) {
|
||||
return 32768
|
||||
} else if (model.includes('claude-opus')) {
|
||||
return 32000
|
||||
} else if (model.includes('gpt-4o') || model.includes('codestral')) {
|
||||
return 16384
|
||||
} else if (model.includes('gpt-4-turbo') || model.includes('gpt-3.5')) {
|
||||
return 4096
|
||||
}
|
||||
return 8192
|
||||
// Thinking counts toward max_tokens, so a model outside the table that is asked to
|
||||
// reason gets more room than the plain fallback. A host rejects a budget above the
|
||||
// model's cap, so the reasoning fallback must stay within the cap of every model it
|
||||
// reaches (true of every reasoning model OpenRouter serves outside the table). Custom
|
||||
// AI never gets it: the registry does not know its models, and the per-model override
|
||||
// is its way to a larger budget.
|
||||
const FALLBACK_MAX_TOKENS = 8192
|
||||
const REASONING_FALLBACK_MAX_TOKENS = 32768
|
||||
|
||||
/**
|
||||
* The output budget a request sends when the workspace sets no override. `reasoningEffort`
|
||||
* is the effort the request carries, as `resolveRequestReasoning` resolves it.
|
||||
*/
|
||||
export function getModelMaxTokens(
|
||||
provider: AIProvider,
|
||||
model: string,
|
||||
reasoningEffort?: ReasoningEffort
|
||||
): number {
|
||||
const bareModel = stripLegacyThinkingSuffix(model)
|
||||
return (
|
||||
getKnownModelMaxOutputTokens(provider, bareModel) ??
|
||||
(requestsReasoning(provider, bareModel, reasoningEffort)
|
||||
? REASONING_FALLBACK_MAX_TOKENS
|
||||
: FALLBACK_MAX_TOKENS)
|
||||
)
|
||||
}
|
||||
|
||||
// Resolves the completion token cap for a model: the workspace's per-model
|
||||
@@ -356,8 +348,16 @@ export function getModelMaxTokens(provider: AIProvider, model: string) {
|
||||
// Anthropic request paths so both honor the same limit. `cap` bounds the result
|
||||
// (used by short metadata completions, see METADATA_MAX_TOKENS) — a hard ceiling
|
||||
// that wins over both the workspace override and the default.
|
||||
function resolveMaxTokens(modelProvider: AIProviderModel, cap?: number): number {
|
||||
const defaultMaxTokens = getModelMaxTokens(modelProvider.provider, modelProvider.model)
|
||||
function resolveMaxTokens(
|
||||
modelProvider: AIProviderModel,
|
||||
cap?: number,
|
||||
reasoningEffort?: ReasoningEffort
|
||||
): number {
|
||||
const defaultMaxTokens = getModelMaxTokens(
|
||||
modelProvider.provider,
|
||||
modelProvider.model,
|
||||
reasoningEffort
|
||||
)
|
||||
const modelKey = `${modelProvider.provider}:${modelProvider.model}`
|
||||
let customMaxTokensStore: Record<string, number> | undefined
|
||||
try {
|
||||
@@ -379,9 +379,10 @@ export const METADATA_MAX_TOKENS = 4096
|
||||
function getModelSpecificConfig(
|
||||
modelProvider: AIProviderModel,
|
||||
tools?: OpenAI.Chat.Completions.ChatCompletionTool[],
|
||||
maxTokensCap?: number
|
||||
maxTokensCap?: number,
|
||||
reasoningEffort?: ReasoningEffort
|
||||
) {
|
||||
const maxTokens = resolveMaxTokens(modelProvider, maxTokensCap)
|
||||
const maxTokens = resolveMaxTokens(modelProvider, maxTokensCap, reasoningEffort)
|
||||
if (
|
||||
(modelProvider.provider === 'openai' ||
|
||||
modelProvider.provider === 'azure_openai' ||
|
||||
@@ -903,7 +904,8 @@ export function getProviderAndCompletionConfig<K extends boolean>({
|
||||
tools,
|
||||
forceModelProvider,
|
||||
maxTokensCap,
|
||||
promptCaching
|
||||
promptCaching,
|
||||
reasoningEffort
|
||||
}: {
|
||||
messages: ChatCompletionMessageParam[]
|
||||
stream: K
|
||||
@@ -913,6 +915,9 @@ export function getProviderAndCompletionConfig<K extends boolean>({
|
||||
// Opt-in: a cache write costs more than an uncached read, so it only pays off where
|
||||
// the same prefix is sent again. True for the chat loop, false for one-shot calls.
|
||||
promptCaching?: boolean
|
||||
// The effort the caller will add with `applyReasoningToConfig`. Only sizes the
|
||||
// output budget here: a request that reasons needs room for the thinking.
|
||||
reasoningEffort?: ReasoningEffort
|
||||
}): {
|
||||
provider: AIProvider
|
||||
config: K extends true
|
||||
@@ -929,7 +934,7 @@ export function getProviderAndCompletionConfig<K extends boolean>({
|
||||
provider: modelProvider.provider,
|
||||
config: {
|
||||
...providerConfig,
|
||||
...getModelSpecificConfig(modelProvider, tools, maxTokensCap),
|
||||
...getModelSpecificConfig(modelProvider, tools, maxTokensCap, reasoningEffort),
|
||||
messages: processedMessages,
|
||||
stream
|
||||
} as any
|
||||
@@ -1135,7 +1140,8 @@ export async function getCompletion(
|
||||
stream: true,
|
||||
tools,
|
||||
forceModelProvider: options?.forceModelProvider,
|
||||
promptCaching: options?.promptCaching
|
||||
promptCaching: options?.promptCaching,
|
||||
reasoningEffort: options?.reasoningEffort
|
||||
})
|
||||
|
||||
// Use Responses API for OpenAI and Azure OpenAI
|
||||
|
||||
@@ -117,6 +117,58 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
|
||||
['codestral', 32_000]
|
||||
]
|
||||
|
||||
// Output token budgets, matched like the context windows above so one row covers
|
||||
// every host of a model: native API, Azure AI Foundry, OpenRouter, Together, Bedrock.
|
||||
// A host rejects any request above its own limit, so each value is one every host of
|
||||
// the model accepts, not the vendor's maximum. Models not listed fall back in
|
||||
// `getModelMaxTokens`.
|
||||
const MODEL_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [
|
||||
// Anthropic — the Anthropic SDK refuses a non-streaming request whose max_tokens
|
||||
// implies more than ~10 minutes, so these stay below the models' own maximum.
|
||||
// Opus caps at 32K before 4.5.
|
||||
['claude-opus-4.5', 64_000],
|
||||
['claude-opus-4.6', 64_000],
|
||||
['claude-opus-4.7', 64_000],
|
||||
['claude-opus-4.8', 64_000],
|
||||
['claude-opus-5', 64_000],
|
||||
['claude-opus', 32_000],
|
||||
['claude-sonnet', 64_000],
|
||||
['claude-haiku', 64_000],
|
||||
['claude-fable', 64_000],
|
||||
['claude-mythos', 64_000],
|
||||
// OpenAI
|
||||
['gpt-5', 128_000],
|
||||
['gpt-4.1', 32_768],
|
||||
['gpt-4o', 16_384],
|
||||
['gpt-4-turbo', 4_096],
|
||||
['gpt-3.5', 4_096],
|
||||
['o1', 100_000],
|
||||
['o3', 100_000],
|
||||
['o4-mini', 100_000],
|
||||
// Google
|
||||
['gemini-3', 64_000],
|
||||
['gemini-2.5', 64_000]
|
||||
]
|
||||
|
||||
// Open-weight models, which a self-hosted server (Custom AI) caps at whatever context
|
||||
// length its operator set, often below these: vLLM rejects a request whose prompt plus
|
||||
// max_tokens exceeds it. So these rows only apply to the providers that host them.
|
||||
const OPEN_WEIGHT_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [
|
||||
// gpt-oss reasons on every request. Bedrock documents 16K, where Groq takes 65536
|
||||
['gpt-oss', 16_000],
|
||||
// DeepSeek — thinks by default and counts the thinking toward max_tokens. 131072
|
||||
// is DeepSeek's own default at the highest effort. R1 is capped by its OpenRouter
|
||||
// host.
|
||||
['deepseek-v4', 131_072],
|
||||
['deepseek-flash', 131_072],
|
||||
['deepseek-r1', 16_000],
|
||||
// Mistral
|
||||
['codestral', 16_384],
|
||||
// Models named for thinking reason on every request, whether or not the chat sends
|
||||
// an effort. Every one OpenRouter lists takes at least this much.
|
||||
['thinking', 32_768]
|
||||
]
|
||||
|
||||
// Version separators differ by route to the same model: Anthropic writes
|
||||
// `claude-opus-4-8`, OpenRouter writes `anthropic/claude-opus-4.8`. Collapsing
|
||||
// dots to dashes on both sides keeps one table entry covering every route —
|
||||
@@ -197,6 +249,19 @@ export function getKnownModelContextWindow(model: string): number | undefined {
|
||||
return matchModel(MODEL_CONTEXT_WINDOW_MATCHERS, model)
|
||||
}
|
||||
|
||||
const MODEL_MAX_OUTPUT_TOKEN_MATCHERS = buildModelMatchers(MODEL_MAX_OUTPUT_TOKENS)
|
||||
const OPEN_WEIGHT_MAX_OUTPUT_TOKEN_MATCHERS = buildModelMatchers(OPEN_WEIGHT_MAX_OUTPUT_TOKENS)
|
||||
|
||||
export function getKnownModelMaxOutputTokens(
|
||||
provider: AIProvider,
|
||||
model: string
|
||||
): number | undefined {
|
||||
return (
|
||||
matchModel(MODEL_MAX_OUTPUT_TOKEN_MATCHERS, model) ??
|
||||
(provider === 'customai' ? undefined : matchModel(OPEN_WEIGHT_MAX_OUTPUT_TOKEN_MATCHERS, model))
|
||||
)
|
||||
}
|
||||
|
||||
/** Trim/compaction logic needs a number; assume a conservative window when unknown. */
|
||||
export const ASSUMED_CONTEXT_WINDOW = 128000
|
||||
|
||||
|
||||
@@ -374,7 +374,8 @@ const ANTHROPIC_ALWAYS_THINKING = /fable|mythos/
|
||||
/**
|
||||
* Disable token to forward when the user explicitly turns reasoning off on a
|
||||
* model that reasons *by default* — omitting the field would silently keep
|
||||
* the default-on behavior. Undefined means omission is the correct off.
|
||||
* the default-on behavior. Undefined means omission is the correct off. Every
|
||||
* token is `'none'`: `requestsReasoning` reads that value as off.
|
||||
*/
|
||||
export function explicitOffToken(provider: AIProvider, model: string): ReasoningEffort | undefined {
|
||||
switch (reasoningProviderFamily(provider, model)) {
|
||||
@@ -434,6 +435,20 @@ export function resolveRequestReasoning(
|
||||
return undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a request carrying `effort`, as `resolveRequestReasoning` resolves it,
|
||||
* has the model reason. An effort sent to a model the registry does not mark as
|
||||
* reasoning-capable answers no: an explicit choice is always sent, but that does
|
||||
* not make the model think.
|
||||
*/
|
||||
export function requestsReasoning(
|
||||
provider: AIProvider,
|
||||
model: string,
|
||||
effort: ReasoningEffort | undefined
|
||||
): boolean {
|
||||
return !!effort && effort !== 'none' && supportsReasoning(provider, model)
|
||||
}
|
||||
|
||||
export type ReasoningApiKind = 'anthropic' | 'responses' | 'completions' | 'deepseek' | 'mistral'
|
||||
|
||||
/**
|
||||
|
||||
@@ -21,6 +21,7 @@
|
||||
getKnownModelContextWindow,
|
||||
getModelContextWindowFromTable
|
||||
} from '../copilot/modelConfig'
|
||||
import { resolveRequestReasoning } from '../copilot/reasoningRegistry'
|
||||
import { supportsAutocomplete } from '../copilot/utils'
|
||||
import TestAiKey from '../copilot/TestAIKey.svelte'
|
||||
import Label from '../Label.svelte'
|
||||
@@ -650,7 +651,8 @@
|
||||
min={1}
|
||||
max={2_000_000}
|
||||
getDefault={(provider, model) => ({
|
||||
tokens: getModelMaxTokens(provider, model),
|
||||
// What a chat sends at the default effort; the budget depends on it.
|
||||
tokens: getModelMaxTokens(provider, model, resolveRequestReasoning({ provider, model })),
|
||||
assumed: false
|
||||
})}
|
||||
/>
|
||||
|
||||
Reference in New Issue
Block a user