From 546b8d076961398a40e7e53fb1e3e561a39a135b Mon Sep 17 00:00:00 2001 From: Ruben Fiszel Date: Tue, 4 Aug 2026 08:35:56 +0000 Subject: [PATCH] fix: match openrouter model ids by parsed vendor, not raw prefix (#10497) * fix: match openrouter model ids by parsed vendor, not raw prefix Co-Authored-By: Claude Opus 5 (1M context) * fix: anchor context-window matches so a version entry cannot swallow a longer version Co-Authored-By: Claude Opus 5 (1M context) * fix: scope the context-window digit guard to version-suffixed entries Co-Authored-By: Claude Opus 5 (1M context) * docs: state the thinking-suffix invariant without drafting history Co-Authored-By: Claude Opus 5 (1M context) --------- Co-authored-by: Claude Opus 5 (1M context) --- .../copilot/lib.anthropicRouting.test.ts | 40 ++++++----- .../src/lib/components/copilot/lib.test.ts | 16 +++++ .../src/lib/components/copilot/modelConfig.ts | 68 ++++++++++++++++--- .../components/copilot/reasoningRegistry.ts | 5 +- 4 files changed, 100 insertions(+), 29 deletions(-) diff --git a/frontend/src/lib/components/copilot/lib.anthropicRouting.test.ts b/frontend/src/lib/components/copilot/lib.anthropicRouting.test.ts index 014109fe5f..ae2451d49d 100644 --- a/frontend/src/lib/components/copilot/lib.anthropicRouting.test.ts +++ b/frontend/src/lib/components/copilot/lib.anthropicRouting.test.ts @@ -259,25 +259,31 @@ describe('OpenRouter prompt caching', () => { { role: 'tool', tool_call_id: 't1', content: 'tool output' } ] - it('breaks on the system prompt and the newest user turn for Anthropic models', async () => { - const { getCompletion } = await import('./lib') - h.currentModel = { provider: 'openrouter', model: 'anthropic/claude-sonnet-5' } + // The `~` form is OpenRouter's own floating alias for the same vendor, so it has + // to reach the same gate — a prefix match on the raw id silently misses it and + // puts the chat back on full price every turn. + it.each(['anthropic/claude-sonnet-5', '~anthropic/claude-sonnet-latest'])( + 'breaks on the system prompt and the newest user turn for %s', + async (model) => { + const { getCompletion } = await import('./lib') + h.currentModel = { provider: 'openrouter', model } - await getCompletion([...conversation], new AbortController(), undefined, { - promptCaching: true - }) + await getCompletion([...conversation], new AbortController(), undefined, { + promptCaching: true + }) - const sent = openaiCreate.mock.calls[0][0].messages - const ephemeral = { type: 'ephemeral' } - expect(sent[0].content).toEqual([ - { type: 'text', text: 'system prompt', cache_control: ephemeral } - ]) - expect(sent[1].content).toEqual([{ type: 'text', text: 'first', cache_control: ephemeral }]) - expect(sent[3].content).toBe('tool output') - // The chat replays this same history through the Anthropic path, which rejects - // unknown fields on its own blocks, so the originals must come back untouched. - expect(conversation[0].content).toBe('system prompt') - }) + const sent = openaiCreate.mock.calls[0][0].messages + const ephemeral = { type: 'ephemeral' } + expect(sent[0].content).toEqual([ + { type: 'text', text: 'system prompt', cache_control: ephemeral } + ]) + expect(sent[1].content).toEqual([{ type: 'text', text: 'first', cache_control: ephemeral }]) + expect(sent[3].content).toBe('tool output') + // The chat replays this same history through the Anthropic path, which rejects + // unknown fields on its own blocks, so the originals must come back untouched. + expect(conversation[0].content).toBe('system prompt') + } + ) it('leaves other OpenRouter upstreams alone', async () => { const { getCompletion } = await import('./lib') diff --git a/frontend/src/lib/components/copilot/lib.test.ts b/frontend/src/lib/components/copilot/lib.test.ts index 17a7ad6508..3bca2bfab5 100644 --- a/frontend/src/lib/components/copilot/lib.test.ts +++ b/frontend/src/lib/components/copilot/lib.test.ts @@ -217,7 +217,12 @@ describe('model context windows', () => { expect(getKnownModelContextWindow('claude-sonnet-4-6')).toBe(1000000) expect(getKnownModelContextWindow('claude-opus-4-6')).toBe(1000000) expect(getKnownModelContextWindow('claude-opus-4-8')).toBe(1000000) + expect(getKnownModelContextWindow('claude-opus-5')).toBe(1000000) + expect(getKnownModelContextWindow('claude-sonnet-5')).toBe(1000000) expect(getKnownModelContextWindow('anthropic.claude-sonnet-4-6-v1:0')).toBe(1000000) + // OpenRouter dot-versions the same models; both spellings must resolve + expect(getKnownModelContextWindow('anthropic/claude-sonnet-4.6')).toBe(1000000) + expect(getKnownModelContextWindow('anthropic/claude-opus-4.8')).toBe(1000000) }) it('keeps Haiku and older Claude models at 200K', () => { @@ -242,6 +247,17 @@ describe('model context windows', () => { expect(getKnownModelContextWindow('gpt-5.5')).toBe(1000000) }) + it('does not let a version entry claim a longer version number', () => { + // `gpt-4.1` and `gpt-4-1106-preview` collapse to the same prefix, but the + // latter is a 128K model and must stay unlisted + expect(getKnownModelContextWindow('gpt-4-1106-preview')).toBeUndefined() + expect(getKnownModelContextWindow('gpt-4.1-mini')).toBe(1000000) + // family fallbacks still catch a version welded onto the name (Ollama ids), + // which is what keeps compaction on for them + expect(getKnownModelContextWindow('llama3.1')).toBe(128000) + expect(getKnownModelContextWindow('llama3-70b-8192')).toBe(128000) + }) + it('maps recent Gemini and DeepSeek models to the 1M window', () => { expect(getKnownModelContextWindow('gemini-3.1-pro')).toBe(1000000) expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000) diff --git a/frontend/src/lib/components/copilot/modelConfig.ts b/frontend/src/lib/components/copilot/modelConfig.ts index 6805be1a8f..5a0e4a37b3 100644 --- a/frontend/src/lib/components/copilot/modelConfig.ts +++ b/frontend/src/lib/components/copilot/modelConfig.ts @@ -1,5 +1,34 @@ import type { AIProvider } from '$lib/gen' +export type ParsedModelId = { + /** Vendor namespace when the id carries one (`anthropic/claude-sonnet-5`). */ + vendor: string | undefined + /** Bare model id: no vendor prefix, no variant suffix. */ + base: string +} + +/** + * Split a model id into the parts the predicates below match on. Gateways decorate + * the vendor's id in ways a raw substring check misses: OpenRouter marks its own + * floating aliases with a `~` prefix (`~anthropic/claude-sonnet-latest`, distinct + * from the vendor-pinned `anthropic/claude-sonnet-5`) and appends `:variant` + * suffixes (`:free`, `:thinking`). Match on the parsed parts, never on the raw id. + * + * `base` is the last `/` segment: the deprecated `/thinking` selection puts + * its marker where a gateway puts the model, so callers that must resolve one of + * those ids first run it through `stripLegacyThinkingSuffix`. + */ +export function parseModelId(model: string): ParsedModelId { + const normalized = model.toLowerCase().replace(/^~/, '') + const segments = normalized.split('/') + const last = segments[segments.length - 1] + const colon = last.indexOf(':') + return { + vendor: segments.length > 1 ? segments[0] : undefined, + base: colon > 0 ? last.slice(0, colon) : last + } +} + // Azure AI Foundry fronts multiple model families under one resource. Claude // deployments are served only through the Anthropic Messages API, so the chat must // route them like the native Anthropic provider (Anthropic SDK, message format) @@ -19,23 +48,21 @@ export function usesAnthropicMessagesApi(provider: AIProvider, model: string): b // but only documents them for Anthropic-backed models, so the gate is on the routed // model rather than the provider alone. export function usesOpenRouterPromptCaching(provider: AIProvider, model: string): boolean { - return provider === 'openrouter' && model.toLowerCase().startsWith('anthropic/') + return provider === 'openrouter' && parseModelId(model).vendor === 'anthropic' } // gpt-5+ and o-series reasoning models reject the legacy `max_tokens` field on // the OpenAI/Azure Chat Completions API and require `max_completion_tokens` -// instead. The check strips any provider prefix (e.g. OpenRouter's "openai/o3") -// so it matches the bare model id, and the o-series match requires a digit after -// the "o" (o1/o3/o4-mini) so it does not catch unrelated ids like Mistral's -// "open-mistral-*" or "optimus-*". +// instead. The check runs on the bare model id (so OpenRouter's "openai/o3" +// matches), and the o-series match requires a digit after the "o" (o1/o3/o4-mini) +// so it does not catch unrelated ids like Mistral's "open-mistral-*" or "optimus-*". export function requiresMaxCompletionTokens(model: string) { - const normalizedModel = model.toLowerCase() - const baseModel = normalizedModel.split('/').pop() ?? normalizedModel + const baseModel = parseModelId(model).base return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel) } // Context windows of the models we know, most specific entry first — the first -// name included in the model id wins, so provider-prefixed and date-suffixed +// name found in the bare model id wins, so vendor-namespaced and date-suffixed // ids (anthropic.claude-sonnet-4-6-...-v1:0, gpt-5.2-2026-01-01) still resolve. // Conservative family fallbacks sit below the explicit entries; models not // listed at all resolve to undefined, which disables auto-trimming and the @@ -45,6 +72,8 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [ // Haiku, older Claude models (3.x, 4.0, 4.1, 4.5) and date-suffixed Claude 4 // base ids (claude-sonnet-4-20250514) fall through to 200K ['claude-fable-5', 1_000_000], + ['claude-opus-5', 1_000_000], + ['claude-sonnet-5', 1_000_000], ['claude-opus-4-8', 1_000_000], ['claude-opus-4-7', 1_000_000], ['claude-opus-4-6', 1_000_000], @@ -74,8 +103,29 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [ ['codestral', 32_000] ] +// Version separators differ by route to the same model: Anthropic writes +// `claude-opus-4-8`, OpenRouter writes `anthropic/claude-opus-4.8`. Collapsing +// dots to dashes on both sides keeps one table entry covering every route — +// without it a dot-versioned id falls through to a coarser family entry. +function normalizeVersionSeparators(model: string): string { + return model.replace(/\./g, '-') +} + +// An entry that ends on a version digit must not run into a longer version: +// `gpt-4.1` collapses to `gpt-4-1`, which would otherwise claim the 128K +// `gpt-4-1106-preview` as a 1M model. Suffixes that continue with a separator +// (`claude-opus-4-8` in `...-4-8-v1`, `gpt-5` in `gpt-5-mini`) still match. +// Family fallbacks ending on a letter get no such guard — a version welded +// straight onto the name (`llama3.1`) is exactly what they exist to catch. +const MODEL_CONTEXT_WINDOW_MATCHERS: [matcher: RegExp, contextWindow: number][] = + MODEL_CONTEXT_WINDOWS.map(([name, contextWindow]) => { + const pattern = normalizeVersionSeparators(name).replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + return [new RegExp(/\d$/.test(pattern) ? `${pattern}(?!\\d)` : pattern), contextWindow] + }) + export function getKnownModelContextWindow(model: string): number | undefined { - return MODEL_CONTEXT_WINDOWS.find(([name]) => model.includes(name))?.[1] + const id = normalizeVersionSeparators(parseModelId(model).base) + return MODEL_CONTEXT_WINDOW_MATCHERS.find(([matcher]) => matcher.test(id))?.[1] } export function getModelContextWindow(model: string) { diff --git a/frontend/src/lib/components/copilot/reasoningRegistry.ts b/frontend/src/lib/components/copilot/reasoningRegistry.ts index 519c1a4021..581715882d 100644 --- a/frontend/src/lib/components/copilot/reasoningRegistry.ts +++ b/frontend/src/lib/components/copilot/reasoningRegistry.ts @@ -1,5 +1,5 @@ import type { AIProvider, AIProviderModel } from '$lib/gen' -import { usesAnthropicMessagesApi } from './modelConfig' +import { parseModelId, usesAnthropicMessagesApi } from './modelConfig' /** * Reasoning effort is provider/model-specific. We never normalize a single @@ -35,8 +35,7 @@ export function stripLegacyThinkingSuffix(model: string): string { /** Bare model id without any provider/gateway prefix (e.g. OpenRouter's `openai/o3`). */ function baseModelId(model: string): string { - const normalized = model.toLowerCase() - return normalized.split('/').pop() ?? normalized + return parseModelId(model).base } /**