fix: match openrouter model ids by parsed vendor, not raw prefix (#10497)

* fix: match openrouter model ids by parsed vendor, not raw prefix

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* fix: anchor context-window matches so a version entry cannot swallow a longer version

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* fix: scope the context-window digit guard to version-suffixed entries

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* docs: state the thinking-suffix invariant without drafting history

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Ruben Fiszel
2026-08-04 10:35:56 +02:00
committed by GitHub
co-authored by Claude Opus 5
parent 99e7661a1b
commit 546b8d0769
4 changed files with 100 additions and 29 deletions
@@ -259,25 +259,31 @@ describe('OpenRouter prompt caching', () => {
{ role: 'tool', tool_call_id: 't1', content: 'tool output' }
]
it('breaks on the system prompt and the newest user turn for Anthropic models', async () => {
const { getCompletion } = await import('./lib')
h.currentModel = { provider: 'openrouter', model: 'anthropic/claude-sonnet-5' }
// The `~` form is OpenRouter's own floating alias for the same vendor, so it has
// to reach the same gate — a prefix match on the raw id silently misses it and
// puts the chat back on full price every turn.
it.each(['anthropic/claude-sonnet-5', '~anthropic/claude-sonnet-latest'])(
'breaks on the system prompt and the newest user turn for %s',
async (model) => {
const { getCompletion } = await import('./lib')
h.currentModel = { provider: 'openrouter', model }
await getCompletion([...conversation], new AbortController(), undefined, {
promptCaching: true
})
await getCompletion([...conversation], new AbortController(), undefined, {
promptCaching: true
})
const sent = openaiCreate.mock.calls[0][0].messages
const ephemeral = { type: 'ephemeral' }
expect(sent[0].content).toEqual([
{ type: 'text', text: 'system prompt', cache_control: ephemeral }
])
expect(sent[1].content).toEqual([{ type: 'text', text: 'first', cache_control: ephemeral }])
expect(sent[3].content).toBe('tool output')
// The chat replays this same history through the Anthropic path, which rejects
// unknown fields on its own blocks, so the originals must come back untouched.
expect(conversation[0].content).toBe('system prompt')
})
const sent = openaiCreate.mock.calls[0][0].messages
const ephemeral = { type: 'ephemeral' }
expect(sent[0].content).toEqual([
{ type: 'text', text: 'system prompt', cache_control: ephemeral }
])
expect(sent[1].content).toEqual([{ type: 'text', text: 'first', cache_control: ephemeral }])
expect(sent[3].content).toBe('tool output')
// The chat replays this same history through the Anthropic path, which rejects
// unknown fields on its own blocks, so the originals must come back untouched.
expect(conversation[0].content).toBe('system prompt')
}
)
it('leaves other OpenRouter upstreams alone', async () => {
const { getCompletion } = await import('./lib')
@@ -217,7 +217,12 @@ describe('model context windows', () => {
expect(getKnownModelContextWindow('claude-sonnet-4-6')).toBe(1000000)
expect(getKnownModelContextWindow('claude-opus-4-6')).toBe(1000000)
expect(getKnownModelContextWindow('claude-opus-4-8')).toBe(1000000)
expect(getKnownModelContextWindow('claude-opus-5')).toBe(1000000)
expect(getKnownModelContextWindow('claude-sonnet-5')).toBe(1000000)
expect(getKnownModelContextWindow('anthropic.claude-sonnet-4-6-v1:0')).toBe(1000000)
// OpenRouter dot-versions the same models; both spellings must resolve
expect(getKnownModelContextWindow('anthropic/claude-sonnet-4.6')).toBe(1000000)
expect(getKnownModelContextWindow('anthropic/claude-opus-4.8')).toBe(1000000)
})
it('keeps Haiku and older Claude models at 200K', () => {
@@ -242,6 +247,17 @@ describe('model context windows', () => {
expect(getKnownModelContextWindow('gpt-5.5')).toBe(1000000)
})
it('does not let a version entry claim a longer version number', () => {
// `gpt-4.1` and `gpt-4-1106-preview` collapse to the same prefix, but the
// latter is a 128K model and must stay unlisted
expect(getKnownModelContextWindow('gpt-4-1106-preview')).toBeUndefined()
expect(getKnownModelContextWindow('gpt-4.1-mini')).toBe(1000000)
// family fallbacks still catch a version welded onto the name (Ollama ids),
// which is what keeps compaction on for them
expect(getKnownModelContextWindow('llama3.1')).toBe(128000)
expect(getKnownModelContextWindow('llama3-70b-8192')).toBe(128000)
})
it('maps recent Gemini and DeepSeek models to the 1M window', () => {
expect(getKnownModelContextWindow('gemini-3.1-pro')).toBe(1000000)
expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000)
@@ -1,5 +1,34 @@
import type { AIProvider } from '$lib/gen'
export type ParsedModelId = {
/** Vendor namespace when the id carries one (`anthropic/claude-sonnet-5`). */
vendor: string | undefined
/** Bare model id: no vendor prefix, no variant suffix. */
base: string
}
/**
* Split a model id into the parts the predicates below match on. Gateways decorate
* the vendor's id in ways a raw substring check misses: OpenRouter marks its own
* floating aliases with a `~` prefix (`~anthropic/claude-sonnet-latest`, distinct
* from the vendor-pinned `anthropic/claude-sonnet-5`) and appends `:variant`
* suffixes (`:free`, `:thinking`). Match on the parsed parts, never on the raw id.
*
* `base` is the last `/` segment: the deprecated `<model>/thinking` selection puts
* its marker where a gateway puts the model, so callers that must resolve one of
* those ids first run it through `stripLegacyThinkingSuffix`.
*/
export function parseModelId(model: string): ParsedModelId {
const normalized = model.toLowerCase().replace(/^~/, '')
const segments = normalized.split('/')
const last = segments[segments.length - 1]
const colon = last.indexOf(':')
return {
vendor: segments.length > 1 ? segments[0] : undefined,
base: colon > 0 ? last.slice(0, colon) : last
}
}
// Azure AI Foundry fronts multiple model families under one resource. Claude
// deployments are served only through the Anthropic Messages API, so the chat must
// route them like the native Anthropic provider (Anthropic SDK, message format)
@@ -19,23 +48,21 @@ export function usesAnthropicMessagesApi(provider: AIProvider, model: string): b
// but only documents them for Anthropic-backed models, so the gate is on the routed
// model rather than the provider alone.
export function usesOpenRouterPromptCaching(provider: AIProvider, model: string): boolean {
return provider === 'openrouter' && model.toLowerCase().startsWith('anthropic/')
return provider === 'openrouter' && parseModelId(model).vendor === 'anthropic'
}
// gpt-5+ and o-series reasoning models reject the legacy `max_tokens` field on
// the OpenAI/Azure Chat Completions API and require `max_completion_tokens`
// instead. The check strips any provider prefix (e.g. OpenRouter's "openai/o3")
// so it matches the bare model id, and the o-series match requires a digit after
// the "o" (o1/o3/o4-mini) so it does not catch unrelated ids like Mistral's
// "open-mistral-*" or "optimus-*".
// instead. The check runs on the bare model id (so OpenRouter's "openai/o3"
// matches), and the o-series match requires a digit after the "o" (o1/o3/o4-mini)
// so it does not catch unrelated ids like Mistral's "open-mistral-*" or "optimus-*".
export function requiresMaxCompletionTokens(model: string) {
const normalizedModel = model.toLowerCase()
const baseModel = normalizedModel.split('/').pop() ?? normalizedModel
const baseModel = parseModelId(model).base
return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel)
}
// Context windows of the models we know, most specific entry first — the first
// name included in the model id wins, so provider-prefixed and date-suffixed
// name found in the bare model id wins, so vendor-namespaced and date-suffixed
// ids (anthropic.claude-sonnet-4-6-...-v1:0, gpt-5.2-2026-01-01) still resolve.
// Conservative family fallbacks sit below the explicit entries; models not
// listed at all resolve to undefined, which disables auto-trimming and the
@@ -45,6 +72,8 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
// Haiku, older Claude models (3.x, 4.0, 4.1, 4.5) and date-suffixed Claude 4
// base ids (claude-sonnet-4-20250514) fall through to 200K
['claude-fable-5', 1_000_000],
['claude-opus-5', 1_000_000],
['claude-sonnet-5', 1_000_000],
['claude-opus-4-8', 1_000_000],
['claude-opus-4-7', 1_000_000],
['claude-opus-4-6', 1_000_000],
@@ -74,8 +103,29 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
['codestral', 32_000]
]
// Version separators differ by route to the same model: Anthropic writes
// `claude-opus-4-8`, OpenRouter writes `anthropic/claude-opus-4.8`. Collapsing
// dots to dashes on both sides keeps one table entry covering every route —
// without it a dot-versioned id falls through to a coarser family entry.
function normalizeVersionSeparators(model: string): string {
return model.replace(/\./g, '-')
}
// An entry that ends on a version digit must not run into a longer version:
// `gpt-4.1` collapses to `gpt-4-1`, which would otherwise claim the 128K
// `gpt-4-1106-preview` as a 1M model. Suffixes that continue with a separator
// (`claude-opus-4-8` in `...-4-8-v1`, `gpt-5` in `gpt-5-mini`) still match.
// Family fallbacks ending on a letter get no such guard — a version welded
// straight onto the name (`llama3.1`) is exactly what they exist to catch.
const MODEL_CONTEXT_WINDOW_MATCHERS: [matcher: RegExp, contextWindow: number][] =
MODEL_CONTEXT_WINDOWS.map(([name, contextWindow]) => {
const pattern = normalizeVersionSeparators(name).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
return [new RegExp(/\d$/.test(pattern) ? `${pattern}(?!\\d)` : pattern), contextWindow]
})
export function getKnownModelContextWindow(model: string): number | undefined {
return MODEL_CONTEXT_WINDOWS.find(([name]) => model.includes(name))?.[1]
const id = normalizeVersionSeparators(parseModelId(model).base)
return MODEL_CONTEXT_WINDOW_MATCHERS.find(([matcher]) => matcher.test(id))?.[1]
}
export function getModelContextWindow(model: string) {
@@ -1,5 +1,5 @@
import type { AIProvider, AIProviderModel } from '$lib/gen'
import { usesAnthropicMessagesApi } from './modelConfig'
import { parseModelId, usesAnthropicMessagesApi } from './modelConfig'
/**
* Reasoning effort is provider/model-specific. We never normalize a single
@@ -35,8 +35,7 @@ export function stripLegacyThinkingSuffix(model: string): string {
/** Bare model id without any provider/gateway prefix (e.g. OpenRouter's `openai/o3`). */
function baseModelId(model: string): string {
const normalized = model.toLowerCase()
return normalized.split('/').pop() ?? normalized
return parseModelId(model).base
}
/**