From 24f60dccf592938ceec276ef0e39cdc7f6d5d2a7 Mon Sep 17 00:00:00 2001 From: Jinwoo-H Date: Thu, 10 Sep 2026 17:37:27 -0400 Subject: [PATCH] fix(ai-vault-search): preserve search tokens and index Codex tools --- .../session-search-live-transcript.test.ts | 10 +- .../session-search-message-rows.test.ts | 22 ++++ .../session-search-message-rows.ts | 19 +-- .../ai-vault-search/session-search-schema.ts | 2 +- .../session-search-transcript-fixtures.ts | 13 +- .../session-scanner-codex-message-records.ts | 9 ++ .../ai-vault/session-scanner-codex-parser.ts | 7 +- .../session-scanner-codex-record-fast-path.ts | 14 +- ...session-scanner-codex-tool-records.test.ts | 121 ++++++++++++++++++ .../session-scanner-codex-tool-records.ts | 95 ++++++++++++++ 10 files changed, 285 insertions(+), 27 deletions(-) create mode 100644 src/main/ai-vault/session-scanner-codex-tool-records.test.ts create mode 100644 src/main/ai-vault/session-scanner-codex-tool-records.ts diff --git a/src/main/ai-vault-search/session-search-live-transcript.test.ts b/src/main/ai-vault-search/session-search-live-transcript.test.ts index 9d80c8b405b..7f37ca662cb 100644 --- a/src/main/ai-vault-search/session-search-live-transcript.test.ts +++ b/src/main/ai-vault-search/session-search-live-transcript.test.ts @@ -102,8 +102,8 @@ it('keeps a tool result searchable but out of the conversation half', async () = path, `${codexRolloutLines( ['rg', 'pericardium'], - 'src/main/pericardium.ts:12: match', - 'search for the pericardium module' + `outputonly ${'padding '.repeat(600)}tailonly`, + 'promptonly search for the module' ).join('\n')}\n` ) await parseTranscript(path, 'codex', codexHome) @@ -112,7 +112,11 @@ it('keeps a tool result searchable but out of the conversation half', async () = expect(sessionsMatching('pericardium')).toHaveLength(1) // The prompt is conversation; the command output is not, and the column // filter is what tells them apart. - expect(sessionsMatching('{user_text assistant_text}: pericardium')).toHaveLength(1) + expect(sessionsMatching('outputonly')).toHaveLength(1) + expect(sessionsMatching('tailonly')).toHaveLength(0) + expect(sessionsMatching('rg')).toHaveLength(1) + expect(sessionsMatching('{user_text assistant_text}: promptonly')).toHaveLength(1) + expect(sessionsMatching('{user_text assistant_text}: outputonly')).toHaveLength(0) expect(sessionsMatching('{user_text assistant_text}: rg')).toHaveLength(0) }) diff --git a/src/main/ai-vault-search/session-search-message-rows.test.ts b/src/main/ai-vault-search/session-search-message-rows.test.ts index 90044671448..794cfc253ca 100644 --- a/src/main/ai-vault-search/session-search-message-rows.test.ts +++ b/src/main/ai-vault-search/session-search-message-rows.test.ts @@ -58,6 +58,28 @@ it('cuts at whitespace rather than through the word on the boundary', async () = } }) +it.each(['/repo/pericardium.ts', 'PROJ-12345', 'C++', 'cafe\u0301ine'])( + 'preserves the exact FTS token %s at a chunk boundary', + async (token) => { + const index = await openSessionSearchIndexFile('ss-rows-tokenchars') + try { + const text = ' '.repeat(7998) + token + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])] + expect(chunks.map((row) => row.text).join('')).toBe(text) + for (const row of chunks) { + insertSearchMessage(index.db, 1, row) + } + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get(`"${token}"`) + ).toEqual({ n: 1 }) + } finally { + await index.close() + } + } +) + it('backs up to any whitespace, not only a newline', () => { // An ideographic space separates words in a CJK transcript exactly as a // space does here, and a newline-only backoff tears the token after it. diff --git a/src/main/ai-vault-search/session-search-message-rows.ts b/src/main/ai-vault-search/session-search-message-rows.ts index 89bde8872df..98e65d5353c 100644 --- a/src/main/ai-vault-search/session-search-message-rows.ts +++ b/src/main/ai-vault-search/session-search-message-rows.ts @@ -20,23 +20,8 @@ const CHUNK_TARGET_CHARS = 8000 */ const TOOL_ROW_CHARS = 3072 -/** - * A character no FTS token can contain, so a cut just past it tears nothing. - * - * Whitespace alone is not enough. Minified JSON, a base64 blob and a one-line - * log all run past 8,000 characters without a space, so a whitespace-only - * backoff finds nothing and the cut lands inside whatever word straddles the - * target — `pericardium` becomes `perica` in one row and `rdium` in the next, - * and the term the user types matches neither. - * - * Surrogates are excluded so a cut never lands between the two halves of one - * astral character. A few of the tokenizer's `tokenchars` (`. - / +`) are - * treated as boundaries here even though unicode61 keeps them inside a token: - * cutting at one costs the joined form of a path, which is a far smaller loss - * than the torn word this exists to prevent, and only in a window that holds no - * whitespace at all. - */ -const TOKEN_BOUNDARY = /[^\p{L}\p{N}_\uD800-\uDFFF]/u +// Keep unicode61's tokenchars and combining marks intact, including before an available space. +const TOKEN_BOUNDARY = /[^\p{L}\p{N}\p{M}\p{Co}_.\-/+\uD800-\uDFFF]/u /** * Index just past the last token boundary in `[floor, end)`, or -1 when the diff --git a/src/main/ai-vault-search/session-search-schema.ts b/src/main/ai-vault-search/session-search-schema.ts index a53965f7878..2da1e64bc11 100644 --- a/src/main/ai-vault-search/session-search-schema.ts +++ b/src/main/ai-vault-search/session-search-schema.ts @@ -10,7 +10,7 @@ import { removeTreeSync } from '../../shared/windows-transient-lock-removal' // policy, decided where the wire is. // Bump to drop and rebuild: the index is a cache over the transcripts, never a source. -export const SESSION_SEARCH_SCHEMA_VERSION = 4 +export const SESSION_SEARCH_SCHEMA_VERSION = 5 // unicode61 keeps `_ . - /` inside tokens so paths and identifiers match exactly; // the `identifiers` column carries the split form (see session-search-identifier-split). diff --git a/src/main/ai-vault-search/session-search-transcript-fixtures.ts b/src/main/ai-vault-search/session-search-transcript-fixtures.ts index c83b7be07e1..bc7eda9a8ff 100644 --- a/src/main/ai-vault-search/session-search-transcript-fixtures.ts +++ b/src/main/ai-vault-search/session-search-transcript-fixtures.ts @@ -101,11 +101,18 @@ export function codexRolloutLines(command: string[], output: string, prompt: str }), codexLine({ timestamp: recordTimestamp(2), - type: 'event_msg', + type: 'response_item', payload: { - type: 'item_completed', - item: { type: 'CommandExecution', command, aggregated_output: output } + type: 'function_call', + call_id: 'call-1', + name: 'shell', + arguments: JSON.stringify({ command }) } + }), + codexLine({ + timestamp: recordTimestamp(3), + type: 'response_item', + payload: { type: 'function_call_output', call_id: 'call-1', output } }) ] } diff --git a/src/main/ai-vault/session-scanner-codex-message-records.ts b/src/main/ai-vault/session-scanner-codex-message-records.ts index 5aa739275ff..a8a5c3c9d13 100644 --- a/src/main/ai-vault/session-scanner-codex-message-records.ts +++ b/src/main/ai-vault/session-scanner-codex-message-records.ts @@ -1,3 +1,7 @@ +import { + publishCodexResponseTool, + publishCodexCompletedTool +} from './session-scanner-codex-tool-records' import { normalizePromptField } from '../../shared/agent-status-field-normalization' import { addPreviewContent } from './session-scanner-accumulator' import type { SessionAccumulator } from './session-scanner-types' @@ -8,6 +12,10 @@ export function consumeCodexResponseMessage( payload: Record, timestamp: unknown ): boolean { + publishCodexResponseTool(accumulator, payload, timestamp) + if (payload.type !== 'message') { + return false + } accumulator.messageCount++ const role = payload.role === 'assistant' ? 'assistant' : payload.role === 'user' ? 'user' : 'unknown' @@ -24,6 +32,7 @@ export function consumeCodexCompletedMessage( payload: Record, timestamp: unknown ): boolean { + publishCodexCompletedTool(accumulator, payload, timestamp) const item = asRecord(payload.item) if (!item) { return false diff --git a/src/main/ai-vault/session-scanner-codex-parser.ts b/src/main/ai-vault/session-scanner-codex-parser.ts index ef1f3862254..b3571d4a337 100644 --- a/src/main/ai-vault/session-scanner-codex-parser.ts +++ b/src/main/ai-vault/session-scanner-codex-parser.ts @@ -172,7 +172,7 @@ function consumeCodexRecordLine(state: CodexSessionParseState, line: string): vo return } - if (record.type === 'response_item' && payload.type === 'message') { + if (record.type === 'response_item') { if (state.historyMode === 'paginated') { return } @@ -280,7 +280,10 @@ function codexResumeStateFromParseState( return { consumeLine: (line) => consumeCodexRecordLine(state, line), consumeLineBytes: (line) => { - const timelineOnlyRecord = readCodexTimelineOnlyRecord(line) + const timelineOnlyRecord = readCodexTimelineOnlyRecord( + line, + state.accumulator.messages.active && state.historyMode !== 'paginated' + ) if (timelineOnlyRecord) { updateTimeline(state.accumulator, timelineOnlyRecord.timestamp) } else { diff --git a/src/main/ai-vault/session-scanner-codex-record-fast-path.ts b/src/main/ai-vault/session-scanner-codex-record-fast-path.ts index 1c4322f89f6..852ec174e95 100644 --- a/src/main/ai-vault/session-scanner-codex-record-fast-path.ts +++ b/src/main/ai-vault/session-scanner-codex-record-fast-path.ts @@ -1,3 +1,5 @@ +import { CODEX_TOOL_RESPONSE_TYPES } from './session-scanner-codex-tool-records' + // Records below this size are decoded and parsed exactly: JSON.parse on a // kilobyte costs less than the risk of a prefix heuristic, and the scan cost // this path exists to remove is entirely in megabyte-scale records. @@ -22,7 +24,10 @@ const PARSED_EVENT_TYPES = new Set([ ]) /** Returns the timestamp only when the record cannot affect other visible session fields. */ -export function readCodexTimelineOnlyRecord(line: Buffer): { timestamp: string } | null { +export function readCodexTimelineOnlyRecord( + line: Buffer, + includeTools = false +): { timestamp: string } | null { if (line.length <= CODEX_RECORD_PREFIX_LIMIT) { return null } @@ -41,6 +46,13 @@ export function readCodexTimelineOnlyRecord(line: Buffer): { timestamp: string } if (!payloadType) { return null } + if ( + includeTools && + recordType === 'response_item' && + CODEX_TOOL_RESPONSE_TYPES.has(payloadType) + ) { + return null + } const parsedPayloadTypes = recordType === 'response_item' ? PARSED_RESPONSE_ITEM_TYPES : PARSED_EVENT_TYPES return parsedPayloadTypes.has(payloadType) ? null : { timestamp } diff --git a/src/main/ai-vault/session-scanner-codex-tool-records.test.ts b/src/main/ai-vault/session-scanner-codex-tool-records.test.ts new file mode 100644 index 00000000000..75503e6de64 --- /dev/null +++ b/src/main/ai-vault/session-scanner-codex-tool-records.test.ts @@ -0,0 +1,121 @@ +import { expect, it } from 'vitest' +import { createCodexSessionResumeState } from './session-scanner-codex-parser' +import type { TranscriptMessage } from './session-transcript-consumers' +import { readCodexTimelineOnlyRecord } from './session-scanner-codex-record-fast-path' + +const timestamp = '2026-05-01T10:00:00.000Z' +const file = { + path: '/fixture/rollout.jsonl', + mtimeMs: Date.parse(timestamp), + modifiedAt: timestamp +} +const record = (type: string, payload: Record): Buffer => + Buffer.from(JSON.stringify({ timestamp, type, payload })) + +it.each(['function_call_output', 'custom_tool_call_output'])( + 'reads large %s records only when a consumer needs them', + (type) => { + const line = record('response_item', { type, output: 'outputonly '.repeat(300) }) + expect(readCodexTimelineOnlyRecord(line)).toEqual({ timestamp }) + expect(readCodexTimelineOnlyRecord(line, true)).toBeNull() + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!(line) + expect(messages).toEqual([{ role: 'tool', text: 'outputonly '.repeat(300), timestamp }]) + } +) + +it.each([false, true])( + 'uses one tool representation across append when paginated=%s', + async (paginated) => { + const messages: TranscriptMessage[] = [] + let state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + const consume = (type: string, payload: Record) => + state.consumeLineBytes!(record(type, payload)) + consume('session_meta', { id: 'session-1', history_mode: paginated ? 'paginated' : 'full' }) + consume('response_item', { type: 'message', role: 'user', content: 'promptonly' }) + consume('event_msg', { + type: 'item_completed', + item: { type: 'UserMessage', content: [{ type: 'text', text: 'promptonly' }] } + }) + consume('response_item', { + type: 'function_call', + name: 'shell', + arguments: '{"command":"commandonly"}' + }) + // The next scan resumes between the call and its output. + state = state.clone() + consume('response_item', { type: 'function_call_output', output: 'outputonly' }) + consume('event_msg', { + type: 'item_completed', + item: { type: 'CommandExecution', command: ['commandonly'], aggregated_output: 'outputonly' } + }) + expect(messages.filter((message) => message.text.includes('commandonly'))).toHaveLength(1) + expect(messages.filter((message) => message.text === 'outputonly')).toEqual([ + { role: 'tool', text: 'outputonly', timestamp } + ]) + expect(messages.filter((message) => message.role === 'user')).toHaveLength(1) + expect(await state.finalize(process.platform)).toMatchObject({ messageCount: 1 }) + } +) + +it('publishes paginated file changes without counting them as conversation', () => { + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!(record('session_meta', { id: 'session-1', history_mode: 'paginated' })) + state.consumeLineBytes!( + record('event_msg', { + type: 'item_completed', + item: { type: 'FileChange', changes: [{ path: 'src/changed.ts', diff: '+ addedneedle' }] } + }) + ) + expect(messages.map((message) => [message.role, message.text])).toEqual([ + ['tool', 'apply_patch: src/changed.ts'], + ['tool', '+ addedneedle'] + ]) +}) + +it('normalizes custom calls and structured results through the existing content reader', () => { + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!( + record('response_item', { type: 'custom_tool_call', name: 'apply_patch', input: 'patchneedle' }) + ) + state.consumeLineBytes!( + record('response_item', { + type: 'custom_tool_call_output', + output: { content: [{ type: 'text', text: 'resultneedle' }] } + }) + ) + expect(messages.map((message) => message.text)).toEqual([ + 'apply_patch: patchneedle', + 'resultneedle' + ]) +}) + +it('keeps local shell argv searchable', () => { + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!( + record('response_item', { + type: 'local_shell_call', + action: { type: 'exec', command: ['rg', 'argvneedle'] } + }) + ) + expect(messages.map((message) => message.text)).toEqual(['tool: rg argvneedle']) +}) diff --git a/src/main/ai-vault/session-scanner-codex-tool-records.ts b/src/main/ai-vault/session-scanner-codex-tool-records.ts new file mode 100644 index 00000000000..ada7c151478 --- /dev/null +++ b/src/main/ai-vault/session-scanner-codex-tool-records.ts @@ -0,0 +1,95 @@ +import { timestampIso } from './session-scanner-accumulator' +import { asRecord } from './session-scanner-record-value' +import type { SessionAccumulator } from './session-scanner-types' +import { transcriptMessagesFromContent } from './session-transcript-message-content' + +export const CODEX_TOOL_RESPONSE_TYPES = new Set([ + 'function_call', + 'local_shell_call', + 'custom_tool_call', + 'function_call_output', + 'custom_tool_call_output' +]) + +function publishToolContent( + accumulator: SessionAccumulator, + content: unknown, + timestamp: unknown +): void { + for (const message of transcriptMessagesFromContent('tool', content, timestampIso(timestamp))) { + accumulator.messages.push(message) + } +} + +export function publishCodexResponseTool( + accumulator: SessionAccumulator, + payload: Record, + timestamp: unknown +): void { + if (!accumulator.messages.active || !CODEX_TOOL_RESPONSE_TYPES.has(String(payload.type))) { + return + } + if (payload.type === 'function_call_output' || payload.type === 'custom_tool_call_output') { + const output = asRecord(payload.output) + publishToolContent( + accumulator, + [{ type: 'tool_result', content: output?.content ?? output?.output ?? payload.output }], + timestamp + ) + return + } + const input = payload.arguments ?? payload.input ?? payload.action + const action = asRecord(input) + const normalizedInput = + action && Array.isArray(action.command) + ? { ...action, command: action.command.filter((part) => typeof part === 'string').join(' ') } + : input + publishToolContent( + accumulator, + [ + { + type: 'tool_use', + name: payload.name ?? 'tool', + input: normalizedInput + } + ], + timestamp + ) +} + +export function publishCodexCompletedTool( + accumulator: SessionAccumulator, + payload: Record, + timestamp: unknown +): void { + if (!accumulator.messages.active) { + return + } + const item = asRecord(payload.item) + if (item?.type === 'CommandExecution' || item?.type === 'command_execution') { + const command = Array.isArray(item.command) + ? item.command.filter((part) => typeof part === 'string').join(' ') + : item.command + publishToolContent( + accumulator, + [ + { type: 'tool_use', name: 'shell', input: command }, + { type: 'tool_result', content: item.aggregated_output ?? item.aggregatedOutput } + ], + timestamp + ) + } else if (item?.type === 'FileChange' || item?.type === 'file_change') { + const changes = Array.isArray(item.changes) ? item.changes : [] + for (const value of changes) { + const change = asRecord(value) + publishToolContent( + accumulator, + [ + { type: 'tool_use', name: 'apply_patch', input: { path: change?.path } }, + { type: 'tool_result', content: change?.diff } + ], + timestamp + ) + } + } +}