mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-08-22 08:02:19 +00:00
feat(frontend): precise AI chat context usage tracking + indicator (#9551)
* feat: track real ai chat token usage and show context indicator * fix: keep context anchor full-history accurate after trimmed sends * nits * feat: restyle context indicator and disable trim for unknown windows * feat: hide context indicator below 50% usage when window is known * fix: gate 1M claude context window to sonnet/opus 4.6+ * refactor: import context window helpers from modelConfig directly * fix: keep base gpt-5 models at 400k context window * fix: exclude date-suffixed claude 4 ids from 1M window gate * refactor: replace context window heuristics with explicit model table * fix: account for context overhead in trim loop stop condition * fix: re-base context anchor when mode switch changes system prompt or tools * refactor: replace context estimation with usage-report-driven compaction Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * feat: chars/4 fallback for context usage when provider reports none Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix: remove unused slide import failing svelte-check Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix: re-seed context usage estimate on rewind so retry can compact Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * refactor: lazy read-side estimate fallback for context usage Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * test: fix reasoningRegistry mock to match resolveRequestReasoning Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -3,7 +3,8 @@ import { sleep } from '$lib/utils'
|
||||
import { editor as meditor, Position, languages, type IDisposable } from 'monaco-editor'
|
||||
import { LRUCache } from 'lru-cache'
|
||||
import { autocompleteRequest } from './request'
|
||||
import { FIM_MAX_TOKENS, getModelContextWindow } from '../lib'
|
||||
import { FIM_MAX_TOKENS } from '../lib'
|
||||
import { getModelContextWindow } from '../modelConfig'
|
||||
import { setGlobalCSS } from '../shared'
|
||||
import { supportsAutocomplete } from '../utils'
|
||||
import { get } from 'svelte/store'
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
import type { ContextElement } from './context'
|
||||
import ChatQuickActions from './ChatQuickActions.svelte'
|
||||
import ProviderModelSelector from './ProviderModelSelector.svelte'
|
||||
import ContextUsageIndicator from './ContextUsageIndicator.svelte'
|
||||
import AIChatSettingsMenu from './AIChatSettingsMenu.svelte'
|
||||
import ChatMode from './ChatMode.svelte'
|
||||
import DatatableCreationPolicy from './DatatableCreationPolicy.svelte'
|
||||
@@ -535,6 +536,8 @@ the panel, or the Escape-to-stop focus check would wrongly reject them. -->
|
||||
{#if inputPreface}
|
||||
{@render inputPreface()}
|
||||
{/if}
|
||||
<ContextUsageIndicator />
|
||||
|
||||
<AIChatInput
|
||||
bind:this={aiChatInput}
|
||||
bind:selectedContext
|
||||
|
||||
@@ -36,7 +36,8 @@ import { loadApiTools } from './api/apiTools'
|
||||
import { prepareScriptUserMessage } from './script/core'
|
||||
import { prepareNavigatorUserMessage } from './navigator/core'
|
||||
import { sendUserToast } from '$lib/toast'
|
||||
import { getModelContextWindow, workspaceAIClients } from '../lib'
|
||||
import { workspaceAIClients } from '../lib'
|
||||
import { getKnownModelContextWindow } from '../modelConfig'
|
||||
import { dfs } from '$lib/components/flows/previousResults'
|
||||
import { getStringError } from './utils'
|
||||
import { type PasteAttachment } from './pasteTokens'
|
||||
@@ -60,6 +61,7 @@ import type { Selection } from 'monaco-editor'
|
||||
import type AIChatInput from './AIChatInput.svelte'
|
||||
import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core'
|
||||
import { runChatLoop, truncateToToolPairedPrefix } from './chatLoop'
|
||||
import { normalizeContextUsage } from './tokenUsage'
|
||||
import type { ReviewChangesOpts } from './monaco-adapter'
|
||||
import {
|
||||
getCurrentModel,
|
||||
@@ -76,9 +78,15 @@ import {
|
||||
} from './global/core'
|
||||
import { isGlobalAiEnabled } from './global/gate'
|
||||
|
||||
// If the estimated token usage is greater than the model context window - the threshold, we delete the oldest message
|
||||
const MAX_TOKENS_THRESHOLD_PERCENTAGE = 0.05
|
||||
const MAX_TOKENS_HARD_LIMIT = 5000
|
||||
// Drop-oldest compaction of the stored history: once the projected request
|
||||
// size (contextTokens — the provider's report when current, a fresh chars/4
|
||||
// estimate otherwise — plus the new user message) reaches the trigger ratio
|
||||
// of the model's context window, head messages are dropped until roughly the
|
||||
// target ratio. The trigger headroom absorbs what the projection cannot see —
|
||||
// the upcoming completion and tool results, system-prompt/tool-schema changes
|
||||
// from mode switches, and the estimate's chars/4 error.
|
||||
const COMPACTION_TRIGGER_RATIO = 0.8
|
||||
const COMPACTION_TARGET_RATIO = 0.7
|
||||
const AI_AUTONOMY_MODE_STORAGE_KEY = 'ai-chat-autonomy-mode'
|
||||
const LEGACY_AUTO_ACCEPT_TOOL_CONFIRMATIONS_STORAGE_KEY = 'ai-chat-yolo-mode'
|
||||
const WEB_SEARCH_ERROR_HINT =
|
||||
@@ -197,6 +205,12 @@ export class AIChatManager {
|
||||
currentReasoningActive = $state<boolean>(false)
|
||||
displayMessages = $state<DisplayMessage[]>([])
|
||||
messages = $state<ChatCompletionMessageParam[]>([])
|
||||
/** Provider-reported context size of the last committed turn (prompt +
|
||||
* completion of its latest completion — exact, includes system prompt and
|
||||
* tools), or undefined whenever no report describes the current history
|
||||
* (provider never reported, turn failed, history rewound). Never holds a
|
||||
* guess: readers go through `contextTokens`, which estimates lazily. */
|
||||
contextUsage = $state<number | undefined>(undefined)
|
||||
autonomyMode = $state<AIAutonomyMode>(getPersistedAutonomyMode())
|
||||
autoAcceptEditsAvailable = $derived(supportsAutoAcceptEdits(this.mode))
|
||||
autoAcceptEditsActive = $derived(
|
||||
@@ -267,47 +281,89 @@ export class AIChatManager {
|
||||
|
||||
open = $derived(chatState.size > 0)
|
||||
|
||||
checkTokenUsageOverLimit = (messages: ChatCompletionMessageParam[]) => {
|
||||
const estimatedTokens = messages.reduce((acc, message) => {
|
||||
// one token is ~ 4 characters
|
||||
// one token is ~ 4 characters
|
||||
private estimateMessagesTokens = (messages: ChatCompletionMessageParam[]) => {
|
||||
return messages.reduce((acc, message) => {
|
||||
const tokenPerCharacter = 4
|
||||
// handle content
|
||||
if (message.content) {
|
||||
if (typeof message.content === 'string') {
|
||||
acc += message.content.length / tokenPerCharacter
|
||||
} else if (message.content) {
|
||||
acc += JSON.stringify(message.content).length / tokenPerCharacter
|
||||
}
|
||||
// Handle tool calls
|
||||
if (message.role === 'assistant' && message.tool_calls) {
|
||||
acc += JSON.stringify(message.tool_calls).length / tokenPerCharacter
|
||||
}
|
||||
return acc
|
||||
}, 0)
|
||||
const model = getCurrentModel()
|
||||
const modelContextWindow = getModelContextWindow(model.model)
|
||||
return (
|
||||
estimatedTokens >
|
||||
modelContextWindow -
|
||||
Math.max(modelContextWindow * MAX_TOKENS_THRESHOLD_PERCENTAGE, MAX_TOKENS_HARD_LIMIT)
|
||||
)
|
||||
}
|
||||
|
||||
deleteOldestMessage = (messages: ChatCompletionMessageParam[], maxDepth: number = 10) => {
|
||||
if (maxDepth <= 0 || messages.length <= 1) {
|
||||
return messages
|
||||
}
|
||||
const removed = messages.shift()
|
||||
/** Estimated tokens of the parts the messages array doesn't carry: the
|
||||
* current system prompt and tool definitions. */
|
||||
private estimateOverheadTokens = () => {
|
||||
const tokenPerCharacter = 4
|
||||
const systemTokens =
|
||||
typeof this.systemMessage.content === 'string'
|
||||
? this.systemMessage.content.length / tokenPerCharacter
|
||||
: 0
|
||||
const toolTokens =
|
||||
this.tools.length > 0
|
||||
? JSON.stringify(this.tools.map((t) => t.def)).length / tokenPerCharacter
|
||||
: 0
|
||||
return systemTokens + toolTokens
|
||||
}
|
||||
|
||||
// if the removed message is an assistant with tool calls, we need to delete correspding tool response.
|
||||
if (removed?.role === 'assistant' && removed.tool_calls) {
|
||||
if (messages.length > 0 && messages[0]?.role === 'tool') {
|
||||
messages.shift()
|
||||
/**
|
||||
* chars/4 estimate of the full context as currently stored: messages plus
|
||||
* the system prompt and tool definitions the next request would carry.
|
||||
* Recomputed from scratch at each read — never accumulated — so errors
|
||||
* don't compound.
|
||||
*/
|
||||
private estimateWholeContextTokens = () =>
|
||||
Math.round(this.estimateMessagesTokens(this.messages) + this.estimateOverheadTokens())
|
||||
|
||||
/**
|
||||
* How full the context is right now — the single fallback rule, shared by
|
||||
* the compaction trigger and the usage indicator: the provider's exact
|
||||
* report when one describes the current history, a fresh estimate
|
||||
* otherwise. Estimating at the read point (rather than writing estimates
|
||||
* into `contextUsage`) means no code path that mutates history can leave
|
||||
* a stale or missing value behind.
|
||||
*/
|
||||
contextTokens = $derived.by(() => this.contextUsage ?? this.estimateWholeContextTokens())
|
||||
|
||||
/**
|
||||
* Drop-oldest compaction. Deletes messages from the front of the STORED
|
||||
* history (the API messages — displayMessages keep the full conversation
|
||||
* for the user) until at least `tokensToFree` estimated tokens are freed
|
||||
* AND the remaining history starts on a user message: a leading tool
|
||||
* result or assistant turn would dangle without the messages that
|
||||
* introduced it. The most recent user message is never dropped. Returns
|
||||
* the estimated tokens freed.
|
||||
*/
|
||||
compactOldestMessages = (tokensToFree: number): number => {
|
||||
const last = this.messages.length - 1
|
||||
let drop = 0
|
||||
let freed = 0
|
||||
while (drop < last) {
|
||||
if (freed >= tokensToFree && this.messages[drop].role === 'user') {
|
||||
break
|
||||
}
|
||||
freed += this.estimateMessagesTokens([this.messages[drop]])
|
||||
drop++
|
||||
}
|
||||
|
||||
// keep deleting messages until we are under the limit
|
||||
if (this.checkTokenUsageOverLimit(messages)) {
|
||||
return this.deleteOldestMessage(messages, maxDepth - 1)
|
||||
if (drop === 0) {
|
||||
return 0
|
||||
}
|
||||
return messages
|
||||
this.messages = this.messages.slice(drop)
|
||||
// User display messages carry the index of their API message so restart
|
||||
// can rewind to it; re-base them on the compacted history. A message
|
||||
// whose API counterpart was dropped clamps to 0: everything before it
|
||||
// was dropped too (compaction only removes prefixes), so restarting
|
||||
// from it restarts from an empty history.
|
||||
this.displayMessages = this.displayMessages.map((m) =>
|
||||
m.role === 'user' ? { ...m, index: Math.max(0, m.index - drop) } : m
|
||||
)
|
||||
return freed
|
||||
}
|
||||
|
||||
loadApiTools = async () => {
|
||||
@@ -799,7 +855,7 @@ export class AIChatManager {
|
||||
}
|
||||
}
|
||||
})
|
||||
return result.addedMessages
|
||||
return result
|
||||
} catch (err) {
|
||||
console.log('chatRequest error', err)
|
||||
console.error('chatRequest error', err)
|
||||
@@ -1062,18 +1118,43 @@ export class AIChatManager {
|
||||
break
|
||||
}
|
||||
|
||||
// Size of the request about to go out: contextTokens (provider report
|
||||
// when current, fresh chars/4 estimate otherwise) plus the message
|
||||
// being added below. Must be read BEFORE the push — the estimate path
|
||||
// covers the stored history, so pushing first would double-count the
|
||||
// new message.
|
||||
const projectedContextTokens = this.contextTokens + this.estimateMessagesTokens([userMessage])
|
||||
|
||||
this.messages.push(userMessage)
|
||||
const modelLenAfterUser = this.messages.length
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages)
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages, this.contextUsage)
|
||||
|
||||
this.currentReply = ''
|
||||
this.currentReasoning = ''
|
||||
this.currentReasoningActive = false
|
||||
|
||||
let trimmedMessages = [...this.messages]
|
||||
if (this.checkTokenUsageOverLimit(trimmedMessages)) {
|
||||
trimmedMessages = this.deleteOldestMessage(trimmedMessages)
|
||||
// Compaction trigger. Without a known context window there is no limit
|
||||
// to enforce, so compaction stays off rather than guessing one.
|
||||
const contextWindow = model ? getKnownModelContextWindow(model.model) : undefined
|
||||
if (
|
||||
contextWindow !== undefined &&
|
||||
projectedContextTokens >= contextWindow * COMPACTION_TRIGGER_RATIO
|
||||
) {
|
||||
const freed = this.compactOldestMessages(
|
||||
projectedContextTokens - contextWindow * COMPACTION_TARGET_RATIO
|
||||
)
|
||||
// A report stays meaningful only debited by what was dropped; the
|
||||
// estimate path needs no bookkeeping — the next read re-estimates
|
||||
// the compacted history. chars/4 can underestimate the freed
|
||||
// tokens, which errs toward compacting again — never toward
|
||||
// overflowing.
|
||||
if (this.contextUsage !== undefined) {
|
||||
this.contextUsage = Math.max(0, this.contextUsage - freed)
|
||||
}
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages, this.contextUsage)
|
||||
}
|
||||
// Rollback anchor for restoreUnsentTurn: captured after compaction so it
|
||||
// indexes into the (possibly compacted) stored history.
|
||||
const modelLenAfterUser = this.messages.length
|
||||
|
||||
const params: {
|
||||
messages: ChatCompletionMessageParam[]
|
||||
@@ -1083,7 +1164,7 @@ export class AIChatManager {
|
||||
onMessageEnd: () => void
|
||||
}
|
||||
} = {
|
||||
messages: trimmedMessages,
|
||||
messages: [...this.messages],
|
||||
abortController: this.abortController,
|
||||
callbacks: {
|
||||
onNewToken: (token) => (this.currentReply += token),
|
||||
@@ -1159,7 +1240,7 @@ export class AIChatManager {
|
||||
await this.loadApiTools()
|
||||
}
|
||||
|
||||
await this.chatRequest({
|
||||
const result = await this.chatRequest({
|
||||
...params,
|
||||
addedMessages: collectedMessages,
|
||||
onWebSearchUnavailable: () => {
|
||||
@@ -1180,7 +1261,14 @@ export class AIChatManager {
|
||||
if (this.autoAcceptEditsActive) {
|
||||
this.acceptPendingFlowEdits()
|
||||
}
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages)
|
||||
// The report from the last completed iteration still describes the
|
||||
// stored history it was sent with (the kept partial tail is a small
|
||||
// undercount the trigger headroom absorbs). Without one, clear the
|
||||
// stale value — readers estimate via contextTokens.
|
||||
this.contextUsage = result?.lastIterationUsage
|
||||
? result.lastIterationUsage.prompt + result.lastIterationUsage.completion
|
||||
: undefined
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages, this.contextUsage)
|
||||
// Still counts as the saved first turn — skipping the hook here would
|
||||
// permanently miss it (the next turn isn't "first" anymore).
|
||||
if (isFirstUserTurn && this.afterFirstTurnSaved) {
|
||||
@@ -1191,6 +1279,8 @@ export class AIChatManager {
|
||||
} else if (wasAborted || !hasUsableOutput) {
|
||||
// Cancelled before anything usable, or the model returned nothing
|
||||
// (or only reasoning) — treat the turn as unsent (matches Claude Code).
|
||||
// contextUsage is left as-is: the turn is rolled back, so the last
|
||||
// report (pre-turn, possibly debited by compaction) still stands.
|
||||
this.restoreUnsentTurn(displayLenAfterUser, modelLenAfterUser, sentInstructions, sentPastes)
|
||||
if (this.displayMessages.length === 0) {
|
||||
// saveChat no-ops on an empty transcript; the chat persisted earlier
|
||||
@@ -1198,7 +1288,7 @@ export class AIChatManager {
|
||||
// user message on reload. Remove it instead.
|
||||
this.historyManager.deletePastChat(this.historyManager.getCurrentChatId())
|
||||
} else {
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages)
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages, this.contextUsage)
|
||||
}
|
||||
if (!wasAborted) {
|
||||
sendUserToast('The model returned no response — your message was restored to the input.')
|
||||
@@ -1206,10 +1296,18 @@ export class AIChatManager {
|
||||
} else {
|
||||
// Clean turn with output → commit as-is.
|
||||
this.messages = [...this.messages, ...collectedMessages]
|
||||
// The provider's report describes the stored history exactly:
|
||||
// compaction mutates it before sending, so what was sent IS what is
|
||||
// stored — no anchoring or index bookkeeping needed. Without a
|
||||
// report, clear the now-stale value — readers estimate via
|
||||
// contextTokens.
|
||||
this.contextUsage = result?.lastIterationUsage
|
||||
? result.lastIterationUsage.prompt + result.lastIterationUsage.completion
|
||||
: undefined
|
||||
if (this.autoAcceptEditsActive) {
|
||||
this.acceptPendingFlowEdits()
|
||||
}
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages)
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages, this.contextUsage)
|
||||
if (isFirstUserTurn && this.afterFirstTurnSaved) {
|
||||
void Promise.resolve(this.afterFirstTurnSaved()).catch((e) => {
|
||||
console.error('AIChatManager afterFirstTurnSaved hook failed', e)
|
||||
@@ -1223,8 +1321,13 @@ export class AIChatManager {
|
||||
// re-committing would duplicate the turn's messages.
|
||||
if (!turnOutcomeHandled) {
|
||||
this.commitInterruptedTurn(collectedMessages, partialReply)
|
||||
// Any prior report no longer describes the history (a partial turn
|
||||
// was just committed); clear it so readers estimate instead. When
|
||||
// the failure WAS a context-length error, that high estimate forces
|
||||
// compaction on the next send instead of failing the same way again.
|
||||
this.contextUsage = undefined
|
||||
try {
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages)
|
||||
await this.historyManager.saveChat(this.displayMessages, this.messages, this.contextUsage)
|
||||
} catch (saveErr) {
|
||||
console.error('Failed to persist partial chat after error', saveErr)
|
||||
}
|
||||
@@ -1286,6 +1389,12 @@ export class AIChatManager {
|
||||
|
||||
this.messages = this.messages.slice(0, actualMessageIndex)
|
||||
|
||||
// The last report described the pre-rewind history; clear it. Readers
|
||||
// fall back to estimating the rewound history (contextTokens), so the
|
||||
// compaction trigger stays armed — e.g. for Retry after a context-length
|
||||
// error, which rewinds through here.
|
||||
this.contextUsage = undefined
|
||||
|
||||
// Resend the request with the same instructions
|
||||
this.instructions = newContent ?? userMessage.content
|
||||
this.sendRequest({ pastes: pastes ?? userMessage.pastes })
|
||||
@@ -1319,9 +1428,10 @@ export class AIChatManager {
|
||||
|
||||
saveAndClear = async () => {
|
||||
this.cancel('saveAndClear')
|
||||
await this.historyManager.save(this.displayMessages, this.messages)
|
||||
await this.historyManager.save(this.displayMessages, this.messages, this.contextUsage)
|
||||
this.displayMessages = []
|
||||
this.messages = []
|
||||
this.contextUsage = undefined
|
||||
}
|
||||
|
||||
loadPastChat = async (id: string) => {
|
||||
@@ -1329,6 +1439,7 @@ export class AIChatManager {
|
||||
if (chat) {
|
||||
this.displayMessages = chat.displayMessages
|
||||
this.messages = chat.actualMessages
|
||||
this.contextUsage = normalizeContextUsage(chat.contextUsage)
|
||||
this.#automaticScroll = true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -52,7 +52,6 @@ vi.mock('$lib/aiStore', () => ({
|
||||
}))
|
||||
|
||||
vi.mock('../lib', () => ({
|
||||
getModelContextWindow: () => 128000,
|
||||
workspaceAIClients: {
|
||||
subscribe: () => () => undefined,
|
||||
getOpenaiClient: mocks.getOpenaiClient,
|
||||
@@ -311,6 +310,264 @@ describe('AIChatManager persisted autonomy default', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('AIChatManager context compaction', () => {
|
||||
// claude-sonnet-4-6 resolves to a known 1M window (modelConfig is
|
||||
// unmocked): compaction triggers at a projected 800k and drops head
|
||||
// messages until ~700k.
|
||||
const anthropicModel = { provider: 'anthropic', model: 'claude-sonnet-4-6' }
|
||||
|
||||
// The turn-outcome handling rolls back turns with no usable output, so every
|
||||
// sendRequest here must produce a reply to take the clean-commit path.
|
||||
const replyWith = (
|
||||
reply: string,
|
||||
lastIterationUsage: { prompt: number; completion: number; total: number } | null = null
|
||||
) =>
|
||||
mocks.runChatLoop.mockImplementation(async (config: any) => {
|
||||
const message = { role: 'assistant' as const, content: reply }
|
||||
config.addedMessages?.push(message)
|
||||
return {
|
||||
addedMessages: [message],
|
||||
tokenUsage: lastIterationUsage ?? { prompt: 0, completion: 0, total: 0 },
|
||||
lastIterationUsage,
|
||||
hitMaxIterations: false
|
||||
}
|
||||
})
|
||||
|
||||
beforeEach(() => {
|
||||
localStorage.clear()
|
||||
vi.clearAllMocks()
|
||||
mocks.getCurrentModel.mockReturnValue(anthropicModel)
|
||||
mocks.tryGetCurrentModel.mockReturnValue(anthropicModel)
|
||||
replyWith('done')
|
||||
})
|
||||
|
||||
it('compacts the stored history before sending once reported usage projects over the trigger', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400_000) }, // ~100k estimated tokens
|
||||
{ role: 'assistant', content: 'b'.repeat(400_000) }, // ~100k
|
||||
{ role: 'user', content: 'c'.repeat(400) },
|
||||
{ role: 'assistant', content: 'd'.repeat(400) }
|
||||
]
|
||||
// Provider fact: 850k used. Projected past the 800k trigger, so ~150k
|
||||
// must be freed to come back to the 700k target — the first user +
|
||||
// assistant pair (~200k estimated).
|
||||
manager.contextUsage = 850_000
|
||||
manager.instructions = 'next question'
|
||||
const saveChat = vi.spyOn(manager.historyManager, 'saveChat')
|
||||
|
||||
await manager.sendRequest()
|
||||
|
||||
const sent = mocks.runChatLoop.mock.calls[0][0].messages
|
||||
expect(sent.length).toBe(3)
|
||||
expect(sent[0]).toMatchObject({ role: 'user', content: 'c'.repeat(400) })
|
||||
// The mutation is on the stored history, not a per-send copy: the head
|
||||
// pair is gone for good and the turn's reply was committed on top
|
||||
expect(manager.messages.length).toBe(4)
|
||||
expect(manager.messages[0]).toMatchObject({ role: 'user', content: 'c'.repeat(400) })
|
||||
// Mid-turn, the report is debited by the freed estimate (visible in the
|
||||
// compaction-time save) so a rolled-back turn keeps a consistent value
|
||||
expect(saveChat).toHaveBeenCalledWith(expect.anything(), expect.anything(), 650_000)
|
||||
// At commit, the no-report turn clears the stored value; the readable
|
||||
// number falls back to estimating the now-tiny compacted history
|
||||
expect(manager.contextUsage).toBeUndefined()
|
||||
expect(manager.contextTokens).toBeGreaterThan(0)
|
||||
expect(manager.contextTokens).toBeLessThan(50_000)
|
||||
// The display message for the sent prompt re-bases onto the compacted history
|
||||
const userDisplay = manager.displayMessages.find((m) => m.role === 'user')
|
||||
expect(userDisplay && 'index' in userDisplay ? userDisplay.index : undefined).toBe(2)
|
||||
})
|
||||
|
||||
it('updates the reported usage after every send, including compacted ones', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400_000) },
|
||||
{ role: 'assistant', content: 'b'.repeat(400_000) },
|
||||
{ role: 'user', content: 'c'.repeat(400) }
|
||||
]
|
||||
manager.contextUsage = 850_000
|
||||
manager.instructions = 'next question'
|
||||
replyWith('done', { prompt: 720_000, completion: 1_000, total: 721_000 })
|
||||
|
||||
await manager.sendRequest()
|
||||
|
||||
// The report describes exactly what was sent (the compacted history), so
|
||||
// it replaces the debited estimate wholesale.
|
||||
expect(manager.contextUsage).toBe(721_000)
|
||||
})
|
||||
|
||||
it('does not compact while the estimated context stays under the trigger', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400_000) },
|
||||
{ role: 'assistant', content: 'b'.repeat(400_000) }
|
||||
]
|
||||
// no report: the trigger runs off the ~200k estimate, well under 800k
|
||||
manager.instructions = 'next question'
|
||||
|
||||
await manager.sendRequest()
|
||||
|
||||
expect(mocks.runChatLoop.mock.calls[0][0].messages.length).toBe(3)
|
||||
})
|
||||
|
||||
it('compacts off the estimate alone when no report ever arrived', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(1_600_000) }, // ~400k estimated tokens
|
||||
{ role: 'assistant', content: 'b'.repeat(1_600_000) }, // ~400k
|
||||
{ role: 'user', content: 'c'.repeat(400) },
|
||||
{ role: 'assistant', content: 'd'.repeat(400) }
|
||||
]
|
||||
// ~800k estimated with no provider report ever seen (e.g. a gateway that
|
||||
// strips usage): the lazily-estimated projection trips the 800k trigger
|
||||
// and frees down to ~700k — the first user + assistant pair goes
|
||||
manager.instructions = 'next question'
|
||||
|
||||
await manager.sendRequest()
|
||||
|
||||
const sent = mocks.runChatLoop.mock.calls[0][0].messages
|
||||
expect(sent.length).toBe(3)
|
||||
expect(sent[0]).toMatchObject({ role: 'user', content: 'c'.repeat(400) })
|
||||
})
|
||||
|
||||
it('estimates lazily instead of storing a guess when the provider reports no usage', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400_000) },
|
||||
{ role: 'assistant', content: 'b'.repeat(400_000) }
|
||||
]
|
||||
manager.instructions = 'next question'
|
||||
|
||||
await manager.sendRequest() // replyWith('done') reports no usage
|
||||
|
||||
// the stored value stays a pure provider fact…
|
||||
expect(manager.contextUsage).toBeUndefined()
|
||||
// …while the readable number estimates the stored context: ~200k for the
|
||||
// messages plus the real navigator system prompt, tool defs and the small
|
||||
// new-turn messages; the prompt templates aren't pinned here, so assert
|
||||
// the magnitude rather than the byte count
|
||||
expect(manager.contextTokens).toBeGreaterThan(200_000)
|
||||
expect(manager.contextTokens).toBeLessThan(250_000)
|
||||
})
|
||||
|
||||
it('prefers the provider report over the estimate once one arrives', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [{ role: 'user', content: 'a'.repeat(400) }]
|
||||
manager.instructions = 'first'
|
||||
await manager.sendRequest()
|
||||
expect(manager.contextUsage).toBeUndefined()
|
||||
expect(manager.contextTokens).toBeGreaterThan(0)
|
||||
|
||||
replyWith('done', { prompt: 1_234, completion: 56, total: 1_290 })
|
||||
manager.instructions = 'second'
|
||||
await manager.sendRequest()
|
||||
expect(manager.contextUsage).toBe(1_290)
|
||||
expect(manager.contextTokens).toBe(1_290)
|
||||
})
|
||||
|
||||
it('does not compact when the model context window is unknown', async () => {
|
||||
mocks.getCurrentModel.mockReturnValue({ provider: 'custom', model: 'mystery-model-9000' })
|
||||
mocks.tryGetCurrentModel.mockReturnValue({ provider: 'custom', model: 'mystery-model-9000' })
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400_000) },
|
||||
{ role: 'assistant', content: 'b'.repeat(400_000) }
|
||||
]
|
||||
manager.contextUsage = 10_000_000
|
||||
manager.instructions = 'next question'
|
||||
|
||||
await manager.sendRequest()
|
||||
|
||||
expect(mocks.runChatLoop.mock.calls[0][0].messages.length).toBe(3)
|
||||
})
|
||||
|
||||
it('never drops the most recent message', () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [{ role: 'user', content: 'a'.repeat(400_000) }]
|
||||
expect(manager.compactOldestMessages(Number.MAX_SAFE_INTEGER)).toBe(0)
|
||||
expect(manager.messages.length).toBe(1)
|
||||
})
|
||||
|
||||
it('keeps dropping past dangling turns so the history restarts on a user message', () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{
|
||||
role: 'assistant',
|
||||
content: 'calling tools',
|
||||
tool_calls: [
|
||||
{ id: '1', type: 'function', function: { name: 'x', arguments: '{}' } },
|
||||
{ id: '2', type: 'function', function: { name: 'y', arguments: '{}' } }
|
||||
]
|
||||
},
|
||||
{ role: 'tool', content: 'result 1', tool_call_id: '1' },
|
||||
{ role: 'tool', content: 'result 2', tool_call_id: '2' },
|
||||
{ role: 'user', content: 'follow-up' },
|
||||
{ role: 'user', content: 'latest' }
|
||||
]
|
||||
// Freeing 1 token is satisfied by the first drop alone, but the tool
|
||||
// results would dangle without their assistant tool_calls message
|
||||
manager.compactOldestMessages(1)
|
||||
expect(manager.messages.map((m) => m.role)).toEqual(['user', 'user'])
|
||||
})
|
||||
|
||||
it('re-bases display message indices and clamps fully-compacted ones to 0', () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400) }, // ~100 estimated tokens
|
||||
{ role: 'assistant', content: 'b'.repeat(400) }, // ~100
|
||||
{ role: 'user', content: 'c' },
|
||||
{ role: 'user', content: 'd' }
|
||||
]
|
||||
manager.displayMessages = [
|
||||
{ role: 'user', content: 'first', index: 0 },
|
||||
{ role: 'assistant', content: 'answer' },
|
||||
{ role: 'user', content: 'second', index: 2 },
|
||||
{ role: 'user', content: 'third', index: 3 }
|
||||
]
|
||||
manager.compactOldestMessages(150)
|
||||
expect(manager.messages.map((m) => m.content)).toEqual(['c', 'd'])
|
||||
expect(manager.displayMessages.map((m) => ('index' in m ? m.index : undefined))).toEqual([
|
||||
0,
|
||||
undefined,
|
||||
0,
|
||||
1
|
||||
])
|
||||
})
|
||||
|
||||
it('falls back to estimating the rewound history after a rewind', () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.messages = [
|
||||
{ role: 'user', content: 'a'.repeat(400) }, // ~100 estimated tokens
|
||||
{ role: 'assistant', content: 'b'.repeat(400) }, // ~100
|
||||
{ role: 'user', content: 'q2' },
|
||||
{ role: 'assistant', content: 'a2' }
|
||||
]
|
||||
manager.displayMessages = [
|
||||
{ role: 'user', content: 'q1', index: 0 },
|
||||
{ role: 'assistant', content: 'a1' },
|
||||
{ role: 'user', content: 'q2', index: 2 },
|
||||
{ role: 'assistant', content: 'a2' }
|
||||
]
|
||||
// A report that described the pre-rewind history must not survive the
|
||||
// rewind as-is…
|
||||
manager.contextUsage = 999_999
|
||||
manager.restartGeneration(2)
|
||||
expect(manager.contextUsage).toBeUndefined()
|
||||
// …but the readable number stays armed by estimating what remains (the
|
||||
// two surviving messages, plus the prompt/tools the resend installed),
|
||||
// so e.g. Retry after a context-length error still compacts
|
||||
expect(manager.contextTokens).toBeGreaterThanOrEqual(200)
|
||||
expect(manager.contextTokens).toBeLessThan(50_000)
|
||||
})
|
||||
|
||||
it('clears the reported usage when saveAndClear resets the conversation', async () => {
|
||||
const manager = new AIChatManager()
|
||||
manager.contextUsage = 1000
|
||||
await manager.saveAndClear()
|
||||
expect(manager.contextUsage).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
const assistantToolCall = (id: string): ChatCompletionMessageParam => ({
|
||||
role: 'assistant',
|
||||
content: '',
|
||||
@@ -325,8 +582,9 @@ const toolResult = (id: string): ChatCompletionMessageParam => ({
|
||||
describe('AIChatManager sendRequest lifecycle', () => {
|
||||
beforeEach(() => {
|
||||
localStorage.clear()
|
||||
// checkTokenUsageOverLimit reads getCurrentModel().model, so it must be a
|
||||
// real object (the file-level beforeEach defaults it to undefined).
|
||||
// The send path reads the current model (request logging + context window
|
||||
// lookup), so it must be a real object (the file-level beforeEach defaults
|
||||
// it to undefined). 'test-model' has no known window → compaction stays off.
|
||||
mocks.getCurrentModel.mockReturnValue({ model: 'test-model', provider: 'openai' })
|
||||
})
|
||||
|
||||
@@ -340,6 +598,7 @@ describe('AIChatManager sendRequest lifecycle', () => {
|
||||
vi.mocked(runChatLoop).mockResolvedValue({
|
||||
addedMessages: [],
|
||||
tokenUsage: {} as any,
|
||||
lastIterationUsage: null,
|
||||
hitMaxIterations: false
|
||||
})
|
||||
|
||||
@@ -366,7 +625,12 @@ describe('AIChatManager sendRequest lifecycle', () => {
|
||||
config.callbacks.onReasoningStart?.()
|
||||
config.callbacks.onReasoningDelta?.('hmm...')
|
||||
config.callbacks.onMessageEnd()
|
||||
return { addedMessages: [], tokenUsage: {} as any, hitMaxIterations: false }
|
||||
return {
|
||||
addedMessages: [],
|
||||
tokenUsage: {} as any,
|
||||
lastIterationUsage: null,
|
||||
hitMaxIterations: false
|
||||
}
|
||||
})
|
||||
|
||||
manager.instructions = 'do a thing'
|
||||
@@ -387,7 +651,12 @@ describe('AIChatManager sendRequest lifecycle', () => {
|
||||
vi.mocked(runChatLoop).mockImplementation(async (config) => {
|
||||
config.callbacks.onNewToken('hello')
|
||||
config.callbacks.onMessageEnd()
|
||||
return { addedMessages: [], tokenUsage: {} as any, hitMaxIterations: false }
|
||||
return {
|
||||
addedMessages: [],
|
||||
tokenUsage: {} as any,
|
||||
lastIterationUsage: null,
|
||||
hitMaxIterations: false
|
||||
}
|
||||
})
|
||||
|
||||
manager.instructions = 'do a thing'
|
||||
@@ -543,6 +812,7 @@ describe('AIChatManager sendRequest lifecycle', () => {
|
||||
return {
|
||||
addedMessages: config.addedMessages!,
|
||||
tokenUsage: {} as any,
|
||||
lastIterationUsage: null,
|
||||
hitMaxIterations: false
|
||||
}
|
||||
})
|
||||
@@ -573,6 +843,7 @@ describe('AIChatManager sendRequest lifecycle', () => {
|
||||
vi.mocked(runChatLoop).mockResolvedValue({
|
||||
addedMessages: [],
|
||||
tokenUsage: {} as any,
|
||||
lastIterationUsage: null,
|
||||
hitMaxIterations: false
|
||||
})
|
||||
const deletePastChat = vi.spyOn(manager.historyManager, 'deletePastChat')
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
<script lang="ts">
|
||||
import { copilotInfo, copilotSessionModel } from '$lib/aiStore'
|
||||
import { getKnownModelContextWindow } from '../modelConfig'
|
||||
import { getAiChatManager } from './aiChatManagerContext'
|
||||
|
||||
const aiChatManager = getAiChatManager()
|
||||
|
||||
let providerModel = $derived(
|
||||
$copilotSessionModel ?? $copilotInfo.defaultModel ?? $copilotInfo.aiModels[0]
|
||||
)
|
||||
let contextWindow = $derived(
|
||||
providerModel ? getKnownModelContextWindow(providerModel.model) : undefined
|
||||
)
|
||||
// The same number the compaction trigger uses: the provider's report when
|
||||
// one describes the current history (one turn stale by nature), otherwise
|
||||
// a live chars/4 estimate of the stored context.
|
||||
let usedTokens = $derived(Math.round(aiChatManager.contextTokens))
|
||||
// With a known window, only surface once the conversation actually fills it;
|
||||
// without one there is no threshold to compare against, so always show.
|
||||
let visible = $derived(
|
||||
usedTokens > 0 &&
|
||||
aiChatManager.messages.length > 0 &&
|
||||
(contextWindow === undefined || usedTokens >= contextWindow * 0.5)
|
||||
)
|
||||
|
||||
function formatTokenCount(tokens: number): string {
|
||||
if (tokens >= 1_000_000) {
|
||||
return `${(tokens / 1_000_000).toFixed(1).replace(/\.0$/, '')}M`
|
||||
}
|
||||
if (tokens >= 1000) {
|
||||
return `${Math.round(tokens / 1000)}k`
|
||||
}
|
||||
return `${tokens}`
|
||||
}
|
||||
</script>
|
||||
|
||||
{#if visible}
|
||||
<div class="flex justify-end px-1">
|
||||
<span class="text-[0.6rem] text-tertiary tabular-nums" aria-label="Context window usage">
|
||||
context window usage: ~{formatTokenCount(usedTokens)}{contextWindow
|
||||
? ` / ${formatTokenCount(contextWindow)}`
|
||||
: ''}
|
||||
</span>
|
||||
</div>
|
||||
{/if}
|
||||
@@ -3,6 +3,7 @@ import type { DisplayMessage } from './shared'
|
||||
import { expanded, messageDraft } from './chatDraft'
|
||||
import { createLongHash } from '$lib/editorLangUtils'
|
||||
import type { ChatCompletionMessageParam } from 'openai/resources/index.mjs'
|
||||
import type { PersistedContextUsage } from './tokenUsage'
|
||||
interface ChatSchema extends IDBSchema {
|
||||
chats: {
|
||||
key: string
|
||||
@@ -13,6 +14,9 @@ interface ChatSchema extends IDBSchema {
|
||||
title: string
|
||||
lastModified: number
|
||||
sessionId?: string
|
||||
// New writes store the plain reported token count; chats persisted by
|
||||
// earlier versions may still hold the legacy anchor object.
|
||||
contextUsage?: PersistedContextUsage
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -29,6 +33,7 @@ export default class HistoryManager {
|
||||
id: string
|
||||
lastModified: number
|
||||
sessionId?: string
|
||||
contextUsage?: PersistedContextUsage
|
||||
}
|
||||
> = $state({})
|
||||
|
||||
@@ -105,7 +110,11 @@ export default class HistoryManager {
|
||||
return Object.values(this.savedChats)
|
||||
}
|
||||
|
||||
async saveChat(displayMessages: DisplayMessage[], messages: ChatCompletionMessageParam[]) {
|
||||
async saveChat(
|
||||
displayMessages: DisplayMessage[],
|
||||
messages: ChatCompletionMessageParam[],
|
||||
contextUsage?: number
|
||||
) {
|
||||
if (displayMessages.length > 0) {
|
||||
// Expand any collapsed-paste tokens so the title is readable text, not
|
||||
// the chip label + its zero-width id chars.
|
||||
@@ -120,7 +129,8 @@ export default class HistoryManager {
|
||||
title,
|
||||
id: this.currentChatId,
|
||||
lastModified: Date.now(),
|
||||
...(this.sessionId ? { sessionId: this.sessionId } : {})
|
||||
...(this.sessionId ? { sessionId: this.sessionId } : {}),
|
||||
...(contextUsage !== undefined ? { contextUsage } : {})
|
||||
}
|
||||
this.savedChats = {
|
||||
...this.savedChats,
|
||||
@@ -133,8 +143,12 @@ export default class HistoryManager {
|
||||
}
|
||||
}
|
||||
|
||||
async save(displayMessages: DisplayMessage[], messages: ChatCompletionMessageParam[]) {
|
||||
await this.saveChat(displayMessages, messages)
|
||||
async save(
|
||||
displayMessages: DisplayMessage[],
|
||||
messages: ChatCompletionMessageParam[],
|
||||
contextUsage?: number
|
||||
) {
|
||||
await this.saveChat(displayMessages, messages, contextUsage)
|
||||
this.currentChatId = createLongHash()
|
||||
}
|
||||
|
||||
|
||||
@@ -12,7 +12,7 @@ const mocks = vi.hoisted(() => ({
|
||||
parseOpenAIResponsesCompletion: vi.fn(),
|
||||
getAnthropicCompletion: vi.fn(),
|
||||
parseAnthropicCompletion: vi.fn(),
|
||||
resolveEffectiveReasoning: vi.fn()
|
||||
resolveRequestReasoning: vi.fn()
|
||||
}))
|
||||
|
||||
vi.mock('../lib', () => ({
|
||||
@@ -22,7 +22,7 @@ vi.mock('../lib', () => ({
|
||||
}))
|
||||
|
||||
vi.mock('../reasoningRegistry', () => ({
|
||||
resolveEffectiveReasoning: mocks.resolveEffectiveReasoning
|
||||
resolveRequestReasoning: mocks.resolveRequestReasoning
|
||||
}))
|
||||
|
||||
vi.mock('./openai-responses', () => ({
|
||||
@@ -83,7 +83,7 @@ describe('runChatLoop web search fallback', () => {
|
||||
mocks.providerSupportsWebSearch.mockImplementation(
|
||||
(provider) => provider === 'openai' || provider === 'anthropic'
|
||||
)
|
||||
mocks.resolveEffectiveReasoning.mockReturnValue(undefined)
|
||||
mocks.resolveRequestReasoning.mockReturnValue(undefined)
|
||||
mocks.parseOpenAICompletion.mockResolvedValue({
|
||||
shouldContinue: false,
|
||||
tokenUsage
|
||||
@@ -255,9 +255,7 @@ describe('runChatLoop web search fallback', () => {
|
||||
)
|
||||
.mockResolvedValue({})
|
||||
|
||||
await runChatLoop(
|
||||
createConfig({ workspace, callbacks, modelProvider, onWebSearchUnavailable })
|
||||
)
|
||||
await runChatLoop(createConfig({ workspace, callbacks, modelProvider, onWebSearchUnavailable }))
|
||||
|
||||
expect(mocks.getAnthropicCompletion).toHaveBeenCalledTimes(2)
|
||||
expect(mocks.getAnthropicCompletion.mock.calls[0][3]).toEqual(
|
||||
@@ -293,6 +291,46 @@ describe('runChatLoop web search fallback', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('runChatLoop lastIterationUsage', () => {
|
||||
beforeEach(() => {
|
||||
vi.resetAllMocks()
|
||||
mocks.resolveRequestReasoning.mockReturnValue(undefined)
|
||||
})
|
||||
|
||||
it('keeps the usage of the last completion that reported it', async () => {
|
||||
const workspace = `workspace-${randomUUID()}`
|
||||
mocks.getOpenAIResponsesCompletion.mockResolvedValue({})
|
||||
mocks.parseOpenAIResponsesCompletion
|
||||
.mockResolvedValueOnce({
|
||||
shouldContinue: true,
|
||||
tokenUsage: { prompt: 1000, completion: 50, total: 1050 }
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
shouldContinue: false,
|
||||
tokenUsage: { prompt: 1200, completion: 80, total: 1280 }
|
||||
})
|
||||
|
||||
const result = await runChatLoop({ ...createConfig({ workspace }), maxIterations: 2 })
|
||||
|
||||
expect(result.lastIterationUsage).toEqual({ prompt: 1200, completion: 80, total: 1280 })
|
||||
// the aggregate keeps summing across iterations
|
||||
expect(result.tokenUsage).toEqual({ prompt: 2200, completion: 130, total: 2330 })
|
||||
})
|
||||
|
||||
it('ignores empty usage reports and returns null when none are real', async () => {
|
||||
const workspace = `workspace-${randomUUID()}`
|
||||
mocks.getOpenAIResponsesCompletion.mockResolvedValue({})
|
||||
mocks.parseOpenAIResponsesCompletion.mockResolvedValue({
|
||||
shouldContinue: false,
|
||||
tokenUsage: { prompt: 0, completion: 0, total: 0 }
|
||||
})
|
||||
|
||||
const result = await runChatLoop(createConfig({ workspace }))
|
||||
|
||||
expect(result.lastIterationUsage).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
// Builders for the message shapes the chat loop accumulates.
|
||||
const assistant = (content: string): ChatCompletionMessageParam => ({ role: 'assistant', content })
|
||||
const assistantTools = (...ids: string[]): ChatCompletionMessageParam => ({
|
||||
|
||||
@@ -59,7 +59,9 @@ export interface ChatLoopConfig {
|
||||
|
||||
export interface ChatLoopResult {
|
||||
addedMessages: ChatCompletionMessageParam[]
|
||||
/** Sum of usage across all loop iterations (suitable for cost accounting). */
|
||||
tokenUsage: ChatTokenUsage
|
||||
lastIterationUsage: ChatTokenUsage | null
|
||||
hitMaxIterations: boolean
|
||||
}
|
||||
|
||||
@@ -215,9 +217,18 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
|
||||
const addedMessages: ChatCompletionMessageParam[] = config.addedMessages ?? []
|
||||
let tokenUsage = emptyChatTokenUsage()
|
||||
let lastIterationUsage: ChatTokenUsage | null = null
|
||||
let iterations = 0
|
||||
let hitMaxIterations = false
|
||||
|
||||
const trackUsage = (usage: ChatTokenUsage | null | undefined) => {
|
||||
tokenUsage = addChatTokenUsage(tokenUsage, usage)
|
||||
// Some providers/paths report no usage (prompt 0); keep the last real one.
|
||||
if (usage && usage.prompt > 0) {
|
||||
lastIterationUsage = usage
|
||||
}
|
||||
}
|
||||
|
||||
while (true) {
|
||||
if (maxIterations !== undefined && iterations >= maxIterations) {
|
||||
hitMaxIterations = true
|
||||
@@ -283,7 +294,7 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
helpers,
|
||||
parseOptions
|
||||
)
|
||||
tokenUsage = addChatTokenUsage(tokenUsage, continueCompletion.tokenUsage)
|
||||
trackUsage(continueCompletion.tokenUsage)
|
||||
return continueCompletion.shouldContinue
|
||||
}
|
||||
|
||||
@@ -342,7 +353,7 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
undefined,
|
||||
parseOptions
|
||||
)
|
||||
tokenUsage = addChatTokenUsage(tokenUsage, continueCompletion.tokenUsage)
|
||||
trackUsage(continueCompletion.tokenUsage)
|
||||
if (!continueCompletion.shouldContinue) {
|
||||
break
|
||||
}
|
||||
@@ -368,7 +379,7 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
abortController,
|
||||
parseOptions
|
||||
)
|
||||
tokenUsage = addChatTokenUsage(tokenUsage, continueCompletion.tokenUsage)
|
||||
trackUsage(continueCompletion.tokenUsage)
|
||||
return continueCompletion.shouldContinue
|
||||
}
|
||||
|
||||
@@ -403,7 +414,7 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
undefined,
|
||||
parseOptions
|
||||
)
|
||||
tokenUsage = addChatTokenUsage(tokenUsage, continueCompletion.tokenUsage)
|
||||
trackUsage(continueCompletion.tokenUsage)
|
||||
if (!continueCompletion.shouldContinue) {
|
||||
break
|
||||
}
|
||||
@@ -411,5 +422,5 @@ export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResul
|
||||
}
|
||||
}
|
||||
|
||||
return { addedMessages, tokenUsage, hitMaxIterations }
|
||||
return { addedMessages, tokenUsage, lastIterationUsage, hitMaxIterations }
|
||||
}
|
||||
|
||||
@@ -23,7 +23,7 @@ import {
|
||||
} from '../shared'
|
||||
import { createWorkspaceMutationTools } from '../workspaceTools'
|
||||
import { setupTypeAcquisition, type DepsToGet } from '$lib/ata'
|
||||
import { getModelContextWindow } from '../../lib'
|
||||
import { getModelContextWindow } from '../../modelConfig'
|
||||
import type { ReviewChangesOpts } from '../monaco-adapter'
|
||||
import { getCurrentModel } from '$lib/aiStore'
|
||||
import { getDbSchemas } from '$lib/components/apps/components/display/dbtable/metadata'
|
||||
|
||||
@@ -4,6 +4,29 @@ export interface ChatTokenUsage {
|
||||
total: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Context usage persisted by earlier versions, which anchored the provider
|
||||
* report to a message index and re-based it on system-prompt/tool changes.
|
||||
* Usage is now a plain token count; old chats loaded from IndexedDB are
|
||||
* collapsed to it via `normalizeContextUsage`.
|
||||
*/
|
||||
export interface LegacyContextTokenSnapshot {
|
||||
tokens: number
|
||||
atMessageIndex: number
|
||||
overheadEstimate?: number
|
||||
}
|
||||
|
||||
export type PersistedContextUsage = number | LegacyContextTokenSnapshot
|
||||
|
||||
export function normalizeContextUsage(
|
||||
value: PersistedContextUsage | undefined
|
||||
): number | undefined {
|
||||
if (value === undefined) {
|
||||
return undefined
|
||||
}
|
||||
return typeof value === 'number' ? value : value.tokens
|
||||
}
|
||||
|
||||
export function emptyChatTokenUsage(): ChatTokenUsage {
|
||||
return { prompt: 0, completion: 0, total: 0 }
|
||||
}
|
||||
@@ -23,12 +46,17 @@ export function addChatTokenUsage(
|
||||
}
|
||||
}
|
||||
|
||||
export function anthropicUsageToChatTokenUsage(usage: {
|
||||
input_tokens?: number | null
|
||||
output_tokens?: number | null
|
||||
cache_creation_input_tokens?: number | null
|
||||
cache_read_input_tokens?: number | null
|
||||
} | null | undefined): ChatTokenUsage {
|
||||
export function anthropicUsageToChatTokenUsage(
|
||||
usage:
|
||||
| {
|
||||
input_tokens?: number | null
|
||||
output_tokens?: number | null
|
||||
cache_creation_input_tokens?: number | null
|
||||
cache_read_input_tokens?: number | null
|
||||
}
|
||||
| null
|
||||
| undefined
|
||||
): ChatTokenUsage {
|
||||
const prompt =
|
||||
(usage?.input_tokens ?? 0) +
|
||||
(usage?.cache_creation_input_tokens ?? 0) +
|
||||
@@ -42,11 +70,16 @@ export function anthropicUsageToChatTokenUsage(usage: {
|
||||
}
|
||||
}
|
||||
|
||||
export function openAIResponsesUsageToChatTokenUsage(usage: {
|
||||
input_tokens?: number | null
|
||||
output_tokens?: number | null
|
||||
total_tokens?: number | null
|
||||
} | null | undefined): ChatTokenUsage {
|
||||
export function openAIResponsesUsageToChatTokenUsage(
|
||||
usage:
|
||||
| {
|
||||
input_tokens?: number | null
|
||||
output_tokens?: number | null
|
||||
total_tokens?: number | null
|
||||
}
|
||||
| null
|
||||
| undefined
|
||||
): ChatTokenUsage {
|
||||
const prompt = usage?.input_tokens ?? 0
|
||||
const completion = usage?.output_tokens ?? 0
|
||||
|
||||
@@ -57,11 +90,16 @@ export function openAIResponsesUsageToChatTokenUsage(usage: {
|
||||
}
|
||||
}
|
||||
|
||||
export function openAICompletionsUsageToChatTokenUsage(usage: {
|
||||
prompt_tokens?: number | null
|
||||
completion_tokens?: number | null
|
||||
total_tokens?: number | null
|
||||
} | null | undefined): ChatTokenUsage {
|
||||
export function openAICompletionsUsageToChatTokenUsage(
|
||||
usage:
|
||||
| {
|
||||
prompt_tokens?: number | null
|
||||
completion_tokens?: number | null
|
||||
total_tokens?: number | null
|
||||
}
|
||||
| null
|
||||
| undefined
|
||||
): ChatTokenUsage {
|
||||
const prompt = usage?.prompt_tokens ?? 0
|
||||
const completion = usage?.completion_tokens ?? 0
|
||||
|
||||
|
||||
@@ -11,7 +11,11 @@ import {
|
||||
splitContentDelta
|
||||
} from './chat/openaiReasoning'
|
||||
import { parseFimCompletionChoice } from './fim'
|
||||
import { requiresMaxCompletionTokens } from './modelConfig'
|
||||
import {
|
||||
getKnownModelContextWindow,
|
||||
getModelContextWindow,
|
||||
requiresMaxCompletionTokens
|
||||
} from './modelConfig'
|
||||
import { supportsAutocomplete } from './utils'
|
||||
|
||||
type AssistantMessageWithReasoning = ChatCompletionMessageParam & {
|
||||
@@ -206,3 +210,48 @@ describe('splitContentDelta', () => {
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
describe('model context windows', () => {
|
||||
it('maps Sonnet/Opus 4.6+ Claude models to the 1M window', () => {
|
||||
expect(getKnownModelContextWindow('claude-sonnet-4-6')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('claude-opus-4-6')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('claude-opus-4-8')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('anthropic.claude-sonnet-4-6-v1:0')).toBe(1000000)
|
||||
})
|
||||
|
||||
it('keeps Haiku and older Claude models at 200K', () => {
|
||||
expect(getKnownModelContextWindow('claude-haiku-4-5')).toBe(200000)
|
||||
expect(getKnownModelContextWindow('global.anthropic.claude-haiku-4-5-20251001-v1:0')).toBe(
|
||||
200000
|
||||
)
|
||||
expect(getKnownModelContextWindow('claude-3-5-sonnet-latest')).toBe(200000)
|
||||
expect(getKnownModelContextWindow('claude-sonnet-4-5-20250929')).toBe(200000)
|
||||
expect(getKnownModelContextWindow('claude-opus-4-1')).toBe(200000)
|
||||
// date-suffixed base ids without a minor version: the date must not be
|
||||
// captured as the version
|
||||
expect(getKnownModelContextWindow('claude-sonnet-4-20250514')).toBe(200000)
|
||||
expect(getKnownModelContextWindow('anthropic.claude-sonnet-4-20250514-v1:0')).toBe(200000)
|
||||
})
|
||||
|
||||
it('keeps base GPT-5 models at 400K while GPT-5.4+ get the 1M window', () => {
|
||||
expect(getKnownModelContextWindow('gpt-5')).toBe(400000)
|
||||
expect(getKnownModelContextWindow('gpt-5-mini')).toBe(400000)
|
||||
expect(getKnownModelContextWindow('gpt-5.2')).toBe(400000)
|
||||
expect(getKnownModelContextWindow('gpt-5.4')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('gpt-5.5')).toBe(1000000)
|
||||
})
|
||||
|
||||
it('maps recent Gemini and DeepSeek models to the 1M window', () => {
|
||||
expect(getKnownModelContextWindow('gemini-3.1-pro')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('gemini-3-flash')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('gemini-2.5-flash')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-v4-pro')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-chat')).toBe(1000000)
|
||||
expect(getKnownModelContextWindow('deepseek-reasoner')).toBe(1000000)
|
||||
})
|
||||
|
||||
it('returns undefined for unrecognized models, 128K via the defaulting wrapper', () => {
|
||||
expect(getKnownModelContextWindow('some-custom-model')).toBeUndefined()
|
||||
expect(getModelContextWindow('some-custom-model')).toBe(128000)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -287,21 +287,6 @@ export function getModelMaxTokens(provider: AIProvider, model: string) {
|
||||
return 8192
|
||||
}
|
||||
|
||||
export function getModelContextWindow(model: string) {
|
||||
if (model.includes('gpt-4.1') || model.includes('gemini')) {
|
||||
return 1000000
|
||||
} else if (model.includes('gpt-5')) {
|
||||
return 400000
|
||||
} else if (model.includes('gpt-4o') || model.includes('llama-3.3')) {
|
||||
return 128000
|
||||
} else if (model.includes('claude') || model.includes('o4-mini') || model.includes('o3')) {
|
||||
return 200000
|
||||
} else if (model.includes('codestral')) {
|
||||
return 32000
|
||||
} else {
|
||||
return 128000
|
||||
}
|
||||
}
|
||||
|
||||
function getModelSpecificConfig(
|
||||
modelProvider: AIProviderModel,
|
||||
|
||||
@@ -9,3 +9,52 @@ export function requiresMaxCompletionTokens(model: string) {
|
||||
const baseModel = normalizedModel.split('/').pop() ?? normalizedModel
|
||||
return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel)
|
||||
}
|
||||
|
||||
// Context windows of the models we know, most specific entry first — the first
|
||||
// name included in the model id wins, so provider-prefixed and date-suffixed
|
||||
// ids (anthropic.claude-sonnet-4-6-...-v1:0, gpt-5.2-2026-01-01) still resolve.
|
||||
// Conservative family fallbacks sit below the explicit entries; models not
|
||||
// listed at all resolve to undefined, which disables auto-trimming and the
|
||||
// indicator denominator.
|
||||
const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
|
||||
// Anthropic — Sonnet/Opus 4.6+ ship a 1M window at standard pricing (GA);
|
||||
// Haiku, older Claude models (3.x, 4.0, 4.1, 4.5) and date-suffixed Claude 4
|
||||
// base ids (claude-sonnet-4-20250514) fall through to 200K
|
||||
['claude-fable-5', 1_000_000],
|
||||
['claude-opus-4-8', 1_000_000],
|
||||
['claude-opus-4-7', 1_000_000],
|
||||
['claude-opus-4-6', 1_000_000],
|
||||
['claude-sonnet-4-6', 1_000_000],
|
||||
['claude', 200_000],
|
||||
// OpenAI — gpt-5 covers the base family (-mini / -nano) and the 5.1/5.2
|
||||
// revisions, all 400K; only 5.4+ moved to 1M
|
||||
['gpt-5.5', 1_000_000],
|
||||
['gpt-5.4', 1_000_000],
|
||||
['gpt-5', 400_000],
|
||||
['gpt-4.1', 1_000_000],
|
||||
['gpt-4o', 128_000],
|
||||
['o4-mini', 200_000],
|
||||
['o3', 200_000],
|
||||
// Google — the 2.5 / 3 / 3.1 Gemini families are all 1M
|
||||
['gemini-3.1', 1_000_000],
|
||||
['gemini-3', 1_000_000],
|
||||
['gemini-2.5', 1_000_000],
|
||||
// DeepSeek — the V4 family is 1M; deepseek-chat / deepseek-reasoner are
|
||||
// aliases of V4-Flash since April 2026
|
||||
['deepseek-v4', 1_000_000],
|
||||
['deepseek-chat', 1_000_000],
|
||||
['deepseek-reasoner', 1_000_000],
|
||||
['deepseek', 128_000],
|
||||
// Others
|
||||
['llama', 128_000],
|
||||
['codestral', 32_000]
|
||||
]
|
||||
|
||||
export function getKnownModelContextWindow(model: string): number | undefined {
|
||||
return MODEL_CONTEXT_WINDOWS.find(([name]) => model.includes(name))?.[1]
|
||||
}
|
||||
|
||||
export function getModelContextWindow(model: string) {
|
||||
// Trim/compaction logic needs a number; assume a conservative window when unknown.
|
||||
return getKnownModelContextWindow(model) ?? 128000
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user