mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-10-05 08:02:28 +00:00
fix: match openrouter model ids by parsed vendor, not raw prefix (#10497)
* fix: match openrouter model ids by parsed vendor, not raw prefix Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * fix: anchor context-window matches so a version entry cannot swallow a longer version Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * fix: scope the context-window digit guard to version-suffixed entries Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * docs: state the thinking-suffix invariant without drafting history Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
99e7661a1b
commit
546b8d0769
@@ -259,25 +259,31 @@ describe('OpenRouter prompt caching', () => {
|
||||
{ role: 'tool', tool_call_id: 't1', content: 'tool output' }
|
||||
]
|
||||
|
||||
it('breaks on the system prompt and the newest user turn for Anthropic models', async () => {
|
||||
const { getCompletion } = await import('./lib')
|
||||
h.currentModel = { provider: 'openrouter', model: 'anthropic/claude-sonnet-5' }
|
||||
// The `~` form is OpenRouter's own floating alias for the same vendor, so it has
|
||||
// to reach the same gate — a prefix match on the raw id silently misses it and
|
||||
// puts the chat back on full price every turn.
|
||||
it.each(['anthropic/claude-sonnet-5', '~anthropic/claude-sonnet-latest'])(
|
||||
'breaks on the system prompt and the newest user turn for %s',
|
||||
async (model) => {
|
||||
const { getCompletion } = await import('./lib')
|
||||
h.currentModel = { provider: 'openrouter', model }
|
||||
|
||||
await getCompletion([...conversation], new AbortController(), undefined, {
|
||||
promptCaching: true
|
||||
})
|
||||
await getCompletion([...conversation], new AbortController(), undefined, {
|
||||
promptCaching: true
|
||||
})
|
||||
|
||||
const sent = openaiCreate.mock.calls[0][0].messages
|
||||
const ephemeral = { type: 'ephemeral' }
|
||||
expect(sent[0].content).toEqual([
|
||||
{ type: 'text', text: 'system prompt', cache_control: ephemeral }
|
||||
])
|
||||
expect(sent[1].content).toEqual([{ type: 'text', text: 'first', cache_control: ephemeral }])
|
||||
expect(sent[3].content).toBe('tool output')
|
||||
// The chat replays this same history through the Anthropic path, which rejects
|
||||
// unknown fields on its own blocks, so the originals must come back untouched.
|
||||
expect(conversation[0].content).toBe('system prompt')
|
||||
})
|
||||
const sent = openaiCreate.mock.calls[0][0].messages
|
||||
const ephemeral = { type: 'ephemeral' }
|
||||
expect(sent[0].content).toEqual([
|
||||
{ type: 'text', text: 'system prompt', cache_control: ephemeral }
|
||||
])
|
||||
expect(sent[1].content).toEqual([{ type: 'text', text: 'first', cache_control: ephemeral }])
|
||||
expect(sent[3].content).toBe('tool output')
|
||||
// The chat replays this same history through the Anthropic path, which rejects
|
||||
// unknown fields on its own blocks, so the originals must come back untouched.
|
||||
expect(conversation[0].content).toBe('system prompt')
|
||||
}
|
||||
)
|
||||
|
||||
it('leaves other OpenRouter upstreams alone', async () => {
|
||||
const { getCompletion } = await import('./lib')
|
||||
|
||||
@@ -217,7 +217,12 @@ describe('model context windows', () => {
|
||||
expect(getKnownModelContextWindow('claude-sonnet-4-6')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('claude-opus-4-6')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('claude-opus-4-8')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('claude-opus-5')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('claude-sonnet-5')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('anthropic.claude-sonnet-4-6-v1:0')).toBe(1000000)
|
||||
// OpenRouter dot-versions the same models; both spellings must resolve
|
||||
expect(getKnownModelContextWindow('anthropic/claude-sonnet-4.6')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('anthropic/claude-opus-4.8')).toBe(1000000)
|
||||
})
|
||||
|
||||
it('keeps Haiku and older Claude models at 200K', () => {
|
||||
@@ -242,6 +247,17 @@ describe('model context windows', () => {
|
||||
expect(getKnownModelContextWindow('gpt-5.5')).toBe(1000000)
|
||||
})
|
||||
|
||||
it('does not let a version entry claim a longer version number', () => {
|
||||
// `gpt-4.1` and `gpt-4-1106-preview` collapse to the same prefix, but the
|
||||
// latter is a 128K model and must stay unlisted
|
||||
expect(getKnownModelContextWindow('gpt-4-1106-preview')).toBeUndefined()
|
||||
expect(getKnownModelContextWindow('gpt-4.1-mini')).toBe(1000000)
|
||||
// family fallbacks still catch a version welded onto the name (Ollama ids),
|
||||
// which is what keeps compaction on for them
|
||||
expect(getKnownModelContextWindow('llama3.1')).toBe(128000)
|
||||
expect(getKnownModelContextWindow('llama3-70b-8192')).toBe(128000)
|
||||
})
|
||||
|
||||
it('maps recent Gemini and DeepSeek models to the 1M window', () => {
|
||||
expect(getKnownModelContextWindow('gemini-3.1-pro')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000)
|
||||
|
||||
@@ -1,5 +1,34 @@
|
||||
import type { AIProvider } from '$lib/gen'
|
||||
|
||||
export type ParsedModelId = {
|
||||
/** Vendor namespace when the id carries one (`anthropic/claude-sonnet-5`). */
|
||||
vendor: string | undefined
|
||||
/** Bare model id: no vendor prefix, no variant suffix. */
|
||||
base: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a model id into the parts the predicates below match on. Gateways decorate
|
||||
* the vendor's id in ways a raw substring check misses: OpenRouter marks its own
|
||||
* floating aliases with a `~` prefix (`~anthropic/claude-sonnet-latest`, distinct
|
||||
* from the vendor-pinned `anthropic/claude-sonnet-5`) and appends `:variant`
|
||||
* suffixes (`:free`, `:thinking`). Match on the parsed parts, never on the raw id.
|
||||
*
|
||||
* `base` is the last `/` segment: the deprecated `<model>/thinking` selection puts
|
||||
* its marker where a gateway puts the model, so callers that must resolve one of
|
||||
* those ids first run it through `stripLegacyThinkingSuffix`.
|
||||
*/
|
||||
export function parseModelId(model: string): ParsedModelId {
|
||||
const normalized = model.toLowerCase().replace(/^~/, '')
|
||||
const segments = normalized.split('/')
|
||||
const last = segments[segments.length - 1]
|
||||
const colon = last.indexOf(':')
|
||||
return {
|
||||
vendor: segments.length > 1 ? segments[0] : undefined,
|
||||
base: colon > 0 ? last.slice(0, colon) : last
|
||||
}
|
||||
}
|
||||
|
||||
// Azure AI Foundry fronts multiple model families under one resource. Claude
|
||||
// deployments are served only through the Anthropic Messages API, so the chat must
|
||||
// route them like the native Anthropic provider (Anthropic SDK, message format)
|
||||
@@ -19,23 +48,21 @@ export function usesAnthropicMessagesApi(provider: AIProvider, model: string): b
|
||||
// but only documents them for Anthropic-backed models, so the gate is on the routed
|
||||
// model rather than the provider alone.
|
||||
export function usesOpenRouterPromptCaching(provider: AIProvider, model: string): boolean {
|
||||
return provider === 'openrouter' && model.toLowerCase().startsWith('anthropic/')
|
||||
return provider === 'openrouter' && parseModelId(model).vendor === 'anthropic'
|
||||
}
|
||||
|
||||
// gpt-5+ and o-series reasoning models reject the legacy `max_tokens` field on
|
||||
// the OpenAI/Azure Chat Completions API and require `max_completion_tokens`
|
||||
// instead. The check strips any provider prefix (e.g. OpenRouter's "openai/o3")
|
||||
// so it matches the bare model id, and the o-series match requires a digit after
|
||||
// the "o" (o1/o3/o4-mini) so it does not catch unrelated ids like Mistral's
|
||||
// "open-mistral-*" or "optimus-*".
|
||||
// instead. The check runs on the bare model id (so OpenRouter's "openai/o3"
|
||||
// matches), and the o-series match requires a digit after the "o" (o1/o3/o4-mini)
|
||||
// so it does not catch unrelated ids like Mistral's "open-mistral-*" or "optimus-*".
|
||||
export function requiresMaxCompletionTokens(model: string) {
|
||||
const normalizedModel = model.toLowerCase()
|
||||
const baseModel = normalizedModel.split('/').pop() ?? normalizedModel
|
||||
const baseModel = parseModelId(model).base
|
||||
return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel)
|
||||
}
|
||||
|
||||
// Context windows of the models we know, most specific entry first — the first
|
||||
// name included in the model id wins, so provider-prefixed and date-suffixed
|
||||
// name found in the bare model id wins, so vendor-namespaced and date-suffixed
|
||||
// ids (anthropic.claude-sonnet-4-6-...-v1:0, gpt-5.2-2026-01-01) still resolve.
|
||||
// Conservative family fallbacks sit below the explicit entries; models not
|
||||
// listed at all resolve to undefined, which disables auto-trimming and the
|
||||
@@ -45,6 +72,8 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
|
||||
// Haiku, older Claude models (3.x, 4.0, 4.1, 4.5) and date-suffixed Claude 4
|
||||
// base ids (claude-sonnet-4-20250514) fall through to 200K
|
||||
['claude-fable-5', 1_000_000],
|
||||
['claude-opus-5', 1_000_000],
|
||||
['claude-sonnet-5', 1_000_000],
|
||||
['claude-opus-4-8', 1_000_000],
|
||||
['claude-opus-4-7', 1_000_000],
|
||||
['claude-opus-4-6', 1_000_000],
|
||||
@@ -74,8 +103,29 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
|
||||
['codestral', 32_000]
|
||||
]
|
||||
|
||||
// Version separators differ by route to the same model: Anthropic writes
|
||||
// `claude-opus-4-8`, OpenRouter writes `anthropic/claude-opus-4.8`. Collapsing
|
||||
// dots to dashes on both sides keeps one table entry covering every route —
|
||||
// without it a dot-versioned id falls through to a coarser family entry.
|
||||
function normalizeVersionSeparators(model: string): string {
|
||||
return model.replace(/\./g, '-')
|
||||
}
|
||||
|
||||
// An entry that ends on a version digit must not run into a longer version:
|
||||
// `gpt-4.1` collapses to `gpt-4-1`, which would otherwise claim the 128K
|
||||
// `gpt-4-1106-preview` as a 1M model. Suffixes that continue with a separator
|
||||
// (`claude-opus-4-8` in `...-4-8-v1`, `gpt-5` in `gpt-5-mini`) still match.
|
||||
// Family fallbacks ending on a letter get no such guard — a version welded
|
||||
// straight onto the name (`llama3.1`) is exactly what they exist to catch.
|
||||
const MODEL_CONTEXT_WINDOW_MATCHERS: [matcher: RegExp, contextWindow: number][] =
|
||||
MODEL_CONTEXT_WINDOWS.map(([name, contextWindow]) => {
|
||||
const pattern = normalizeVersionSeparators(name).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
||||
return [new RegExp(/\d$/.test(pattern) ? `${pattern}(?!\\d)` : pattern), contextWindow]
|
||||
})
|
||||
|
||||
export function getKnownModelContextWindow(model: string): number | undefined {
|
||||
return MODEL_CONTEXT_WINDOWS.find(([name]) => model.includes(name))?.[1]
|
||||
const id = normalizeVersionSeparators(parseModelId(model).base)
|
||||
return MODEL_CONTEXT_WINDOW_MATCHERS.find(([matcher]) => matcher.test(id))?.[1]
|
||||
}
|
||||
|
||||
export function getModelContextWindow(model: string) {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { AIProvider, AIProviderModel } from '$lib/gen'
|
||||
import { usesAnthropicMessagesApi } from './modelConfig'
|
||||
import { parseModelId, usesAnthropicMessagesApi } from './modelConfig'
|
||||
|
||||
/**
|
||||
* Reasoning effort is provider/model-specific. We never normalize a single
|
||||
@@ -35,8 +35,7 @@ export function stripLegacyThinkingSuffix(model: string): string {
|
||||
|
||||
/** Bare model id without any provider/gateway prefix (e.g. OpenRouter's `openai/o3`). */
|
||||
function baseModelId(model: string): string {
|
||||
const normalized = model.toLowerCase()
|
||||
return normalized.split('/').pop() ?? normalized
|
||||
return parseModelId(model).base
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user