fix(ai-vault-search): preserve search tokens and index Codex tools

This commit is contained in:
Jinwoo-H
2026-09-10 17:37:27 -04:00
parent 724f84b1b7
commit 24f60dccf5
10 changed files with 285 additions and 27 deletions
@@ -102,8 +102,8 @@ it('keeps a tool result searchable but out of the conversation half', async () =
path,
`${codexRolloutLines(
['rg', 'pericardium'],
'src/main/pericardium.ts:12: match',
'search for the pericardium module'
`outputonly ${'padding '.repeat(600)}tailonly`,
'promptonly search for the module'
).join('\n')}\n`
)
await parseTranscript(path, 'codex', codexHome)
@@ -112,7 +112,11 @@ it('keeps a tool result searchable but out of the conversation half', async () =
expect(sessionsMatching('pericardium')).toHaveLength(1)
// The prompt is conversation; the command output is not, and the column
// filter is what tells them apart.
expect(sessionsMatching('{user_text assistant_text}: pericardium')).toHaveLength(1)
expect(sessionsMatching('outputonly')).toHaveLength(1)
expect(sessionsMatching('tailonly')).toHaveLength(0)
expect(sessionsMatching('rg')).toHaveLength(1)
expect(sessionsMatching('{user_text assistant_text}: promptonly')).toHaveLength(1)
expect(sessionsMatching('{user_text assistant_text}: outputonly')).toHaveLength(0)
expect(sessionsMatching('{user_text assistant_text}: rg')).toHaveLength(0)
})
@@ -58,6 +58,28 @@ it('cuts at whitespace rather than through the word on the boundary', async () =
}
})
it.each(['/repo/pericardium.ts', 'PROJ-12345', 'C++', 'cafe\u0301ine'])(
'preserves the exact FTS token %s at a chunk boundary',
async (token) => {
const index = await openSessionSearchIndexFile('ss-rows-tokenchars')
try {
const text = ' '.repeat(7998) + token
const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])]
expect(chunks.map((row) => row.text).join('')).toBe(text)
for (const row of chunks) {
insertSearchMessage(index.db, 1, row)
}
expect(
index.db
.prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?')
.get(`"${token}"`)
).toEqual({ n: 1 })
} finally {
await index.close()
}
}
)
it('backs up to any whitespace, not only a newline', () => {
// An ideographic space separates words in a CJK transcript exactly as a
// space does here, and a newline-only backoff tears the token after it.
@@ -20,23 +20,8 @@ const CHUNK_TARGET_CHARS = 8000
*/
const TOOL_ROW_CHARS = 3072
/**
* A character no FTS token can contain, so a cut just past it tears nothing.
*
* Whitespace alone is not enough. Minified JSON, a base64 blob and a one-line
* log all run past 8,000 characters without a space, so a whitespace-only
* backoff finds nothing and the cut lands inside whatever word straddles the
* target — `pericardium` becomes `perica` in one row and `rdium` in the next,
* and the term the user types matches neither.
*
* Surrogates are excluded so a cut never lands between the two halves of one
* astral character. A few of the tokenizer's `tokenchars` (`. - / +`) are
* treated as boundaries here even though unicode61 keeps them inside a token:
* cutting at one costs the joined form of a path, which is a far smaller loss
* than the torn word this exists to prevent, and only in a window that holds no
* whitespace at all.
*/
const TOKEN_BOUNDARY = /[^\p{L}\p{N}_\uD800-\uDFFF]/u
// Keep unicode61's tokenchars and combining marks intact, including before an available space.
const TOKEN_BOUNDARY = /[^\p{L}\p{N}\p{M}\p{Co}_.\-/+\uD800-\uDFFF]/u
/**
* Index just past the last token boundary in `[floor, end)`, or -1 when the
@@ -10,7 +10,7 @@ import { removeTreeSync } from '../../shared/windows-transient-lock-removal'
// policy, decided where the wire is.
// Bump to drop and rebuild: the index is a cache over the transcripts, never a source.
export const SESSION_SEARCH_SCHEMA_VERSION = 4
export const SESSION_SEARCH_SCHEMA_VERSION = 5
// unicode61 keeps `_ . - /` inside tokens so paths and identifiers match exactly;
// the `identifiers` column carries the split form (see session-search-identifier-split).
@@ -101,11 +101,18 @@ export function codexRolloutLines(command: string[], output: string, prompt: str
}),
codexLine({
timestamp: recordTimestamp(2),
type: 'event_msg',
type: 'response_item',
payload: {
type: 'item_completed',
item: { type: 'CommandExecution', command, aggregated_output: output }
type: 'function_call',
call_id: 'call-1',
name: 'shell',
arguments: JSON.stringify({ command })
}
}),
codexLine({
timestamp: recordTimestamp(3),
type: 'response_item',
payload: { type: 'function_call_output', call_id: 'call-1', output }
})
]
}
@@ -1,3 +1,7 @@
import {
publishCodexResponseTool,
publishCodexCompletedTool
} from './session-scanner-codex-tool-records'
import { normalizePromptField } from '../../shared/agent-status-field-normalization'
import { addPreviewContent } from './session-scanner-accumulator'
import type { SessionAccumulator } from './session-scanner-types'
@@ -8,6 +12,10 @@ export function consumeCodexResponseMessage(
payload: Record<string, unknown>,
timestamp: unknown
): boolean {
publishCodexResponseTool(accumulator, payload, timestamp)
if (payload.type !== 'message') {
return false
}
accumulator.messageCount++
const role =
payload.role === 'assistant' ? 'assistant' : payload.role === 'user' ? 'user' : 'unknown'
@@ -24,6 +32,7 @@ export function consumeCodexCompletedMessage(
payload: Record<string, unknown>,
timestamp: unknown
): boolean {
publishCodexCompletedTool(accumulator, payload, timestamp)
const item = asRecord(payload.item)
if (!item) {
return false
@@ -172,7 +172,7 @@ function consumeCodexRecordLine(state: CodexSessionParseState, line: string): vo
return
}
if (record.type === 'response_item' && payload.type === 'message') {
if (record.type === 'response_item') {
if (state.historyMode === 'paginated') {
return
}
@@ -280,7 +280,10 @@ function codexResumeStateFromParseState(
return {
consumeLine: (line) => consumeCodexRecordLine(state, line),
consumeLineBytes: (line) => {
const timelineOnlyRecord = readCodexTimelineOnlyRecord(line)
const timelineOnlyRecord = readCodexTimelineOnlyRecord(
line,
state.accumulator.messages.active && state.historyMode !== 'paginated'
)
if (timelineOnlyRecord) {
updateTimeline(state.accumulator, timelineOnlyRecord.timestamp)
} else {
@@ -1,3 +1,5 @@
import { CODEX_TOOL_RESPONSE_TYPES } from './session-scanner-codex-tool-records'
// Records below this size are decoded and parsed exactly: JSON.parse on a
// kilobyte costs less than the risk of a prefix heuristic, and the scan cost
// this path exists to remove is entirely in megabyte-scale records.
@@ -22,7 +24,10 @@ const PARSED_EVENT_TYPES = new Set([
])
/** Returns the timestamp only when the record cannot affect other visible session fields. */
export function readCodexTimelineOnlyRecord(line: Buffer): { timestamp: string } | null {
export function readCodexTimelineOnlyRecord(
line: Buffer,
includeTools = false
): { timestamp: string } | null {
if (line.length <= CODEX_RECORD_PREFIX_LIMIT) {
return null
}
@@ -41,6 +46,13 @@ export function readCodexTimelineOnlyRecord(line: Buffer): { timestamp: string }
if (!payloadType) {
return null
}
if (
includeTools &&
recordType === 'response_item' &&
CODEX_TOOL_RESPONSE_TYPES.has(payloadType)
) {
return null
}
const parsedPayloadTypes =
recordType === 'response_item' ? PARSED_RESPONSE_ITEM_TYPES : PARSED_EVENT_TYPES
return parsedPayloadTypes.has(payloadType) ? null : { timestamp }
@@ -0,0 +1,121 @@
import { expect, it } from 'vitest'
import { createCodexSessionResumeState } from './session-scanner-codex-parser'
import type { TranscriptMessage } from './session-transcript-consumers'
import { readCodexTimelineOnlyRecord } from './session-scanner-codex-record-fast-path'
const timestamp = '2026-05-01T10:00:00.000Z'
const file = {
path: '/fixture/rollout.jsonl',
mtimeMs: Date.parse(timestamp),
modifiedAt: timestamp
}
const record = (type: string, payload: Record<string, unknown>): Buffer =>
Buffer.from(JSON.stringify({ timestamp, type, payload }))
it.each(['function_call_output', 'custom_tool_call_output'])(
'reads large %s records only when a consumer needs them',
(type) => {
const line = record('response_item', { type, output: 'outputonly '.repeat(300) })
expect(readCodexTimelineOnlyRecord(line)).toEqual({ timestamp })
expect(readCodexTimelineOnlyRecord(line, true)).toBeNull()
const messages: TranscriptMessage[] = []
const state = createCodexSessionResumeState(file, null, {
active: true,
push: (message) => messages.push(message)
})
state.consumeLineBytes!(line)
expect(messages).toEqual([{ role: 'tool', text: 'outputonly '.repeat(300), timestamp }])
}
)
it.each([false, true])(
'uses one tool representation across append when paginated=%s',
async (paginated) => {
const messages: TranscriptMessage[] = []
let state = createCodexSessionResumeState(file, null, {
active: true,
push: (message) => messages.push(message)
})
const consume = (type: string, payload: Record<string, unknown>) =>
state.consumeLineBytes!(record(type, payload))
consume('session_meta', { id: 'session-1', history_mode: paginated ? 'paginated' : 'full' })
consume('response_item', { type: 'message', role: 'user', content: 'promptonly' })
consume('event_msg', {
type: 'item_completed',
item: { type: 'UserMessage', content: [{ type: 'text', text: 'promptonly' }] }
})
consume('response_item', {
type: 'function_call',
name: 'shell',
arguments: '{"command":"commandonly"}'
})
// The next scan resumes between the call and its output.
state = state.clone()
consume('response_item', { type: 'function_call_output', output: 'outputonly' })
consume('event_msg', {
type: 'item_completed',
item: { type: 'CommandExecution', command: ['commandonly'], aggregated_output: 'outputonly' }
})
expect(messages.filter((message) => message.text.includes('commandonly'))).toHaveLength(1)
expect(messages.filter((message) => message.text === 'outputonly')).toEqual([
{ role: 'tool', text: 'outputonly', timestamp }
])
expect(messages.filter((message) => message.role === 'user')).toHaveLength(1)
expect(await state.finalize(process.platform)).toMatchObject({ messageCount: 1 })
}
)
it('publishes paginated file changes without counting them as conversation', () => {
const messages: TranscriptMessage[] = []
const state = createCodexSessionResumeState(file, null, {
active: true,
push: (message) => messages.push(message)
})
state.consumeLineBytes!(record('session_meta', { id: 'session-1', history_mode: 'paginated' }))
state.consumeLineBytes!(
record('event_msg', {
type: 'item_completed',
item: { type: 'FileChange', changes: [{ path: 'src/changed.ts', diff: '+ addedneedle' }] }
})
)
expect(messages.map((message) => [message.role, message.text])).toEqual([
['tool', 'apply_patch: src/changed.ts'],
['tool', '+ addedneedle']
])
})
it('normalizes custom calls and structured results through the existing content reader', () => {
const messages: TranscriptMessage[] = []
const state = createCodexSessionResumeState(file, null, {
active: true,
push: (message) => messages.push(message)
})
state.consumeLineBytes!(
record('response_item', { type: 'custom_tool_call', name: 'apply_patch', input: 'patchneedle' })
)
state.consumeLineBytes!(
record('response_item', {
type: 'custom_tool_call_output',
output: { content: [{ type: 'text', text: 'resultneedle' }] }
})
)
expect(messages.map((message) => message.text)).toEqual([
'apply_patch: patchneedle',
'resultneedle'
])
})
it('keeps local shell argv searchable', () => {
const messages: TranscriptMessage[] = []
const state = createCodexSessionResumeState(file, null, {
active: true,
push: (message) => messages.push(message)
})
state.consumeLineBytes!(
record('response_item', {
type: 'local_shell_call',
action: { type: 'exec', command: ['rg', 'argvneedle'] }
})
)
expect(messages.map((message) => message.text)).toEqual(['tool: rg argvneedle'])
})
@@ -0,0 +1,95 @@
import { timestampIso } from './session-scanner-accumulator'
import { asRecord } from './session-scanner-record-value'
import type { SessionAccumulator } from './session-scanner-types'
import { transcriptMessagesFromContent } from './session-transcript-message-content'
export const CODEX_TOOL_RESPONSE_TYPES = new Set([
'function_call',
'local_shell_call',
'custom_tool_call',
'function_call_output',
'custom_tool_call_output'
])
function publishToolContent(
accumulator: SessionAccumulator,
content: unknown,
timestamp: unknown
): void {
for (const message of transcriptMessagesFromContent('tool', content, timestampIso(timestamp))) {
accumulator.messages.push(message)
}
}
export function publishCodexResponseTool(
accumulator: SessionAccumulator,
payload: Record<string, unknown>,
timestamp: unknown
): void {
if (!accumulator.messages.active || !CODEX_TOOL_RESPONSE_TYPES.has(String(payload.type))) {
return
}
if (payload.type === 'function_call_output' || payload.type === 'custom_tool_call_output') {
const output = asRecord(payload.output)
publishToolContent(
accumulator,
[{ type: 'tool_result', content: output?.content ?? output?.output ?? payload.output }],
timestamp
)
return
}
const input = payload.arguments ?? payload.input ?? payload.action
const action = asRecord(input)
const normalizedInput =
action && Array.isArray(action.command)
? { ...action, command: action.command.filter((part) => typeof part === 'string').join(' ') }
: input
publishToolContent(
accumulator,
[
{
type: 'tool_use',
name: payload.name ?? 'tool',
input: normalizedInput
}
],
timestamp
)
}
export function publishCodexCompletedTool(
accumulator: SessionAccumulator,
payload: Record<string, unknown>,
timestamp: unknown
): void {
if (!accumulator.messages.active) {
return
}
const item = asRecord(payload.item)
if (item?.type === 'CommandExecution' || item?.type === 'command_execution') {
const command = Array.isArray(item.command)
? item.command.filter((part) => typeof part === 'string').join(' ')
: item.command
publishToolContent(
accumulator,
[
{ type: 'tool_use', name: 'shell', input: command },
{ type: 'tool_result', content: item.aggregated_output ?? item.aggregatedOutput }
],
timestamp
)
} else if (item?.type === 'FileChange' || item?.type === 'file_change') {
const changes = Array.isArray(item.changes) ? item.changes : []
for (const value of changes) {
const change = asRecord(value)
publishToolContent(
accumulator,
[
{ type: 'tool_use', name: 'apply_patch', input: { path: change?.path } },
{ type: 'tool_result', content: change?.diff }
],
timestamp
)
}
}
}