fix(ai-vault-search): tell a highlight from a transcript that contains brackets

The snippet builder asked each of a row's four columns for a marked snippet and
showed the first whose text contained `[[`. Transcripts contain `[[`: a bash
`if [[ -f … ]]`, numpy's `[[1, 2], [3, 4]]`. A row matching only in tool output
was shown its user turn instead, with nothing highlighted in it, and the
any-column fallback an identifier-only match depends on was unreachable behind
the same collision.

Whether a column matched is now the difference between two renderings of the
same text: `snippet(…, MARK, MARK, …)` beside `snippet(…, '', '', …)`. Content
cannot forge a difference between those two, because it is the same content
either way.

The marks FTS5 inserts are private-use code points, rewritten to the public
`[[` and `]]` once, at the end. That is for the other decision that has to tell
a mark from content: the truncation refuses to cut between an open mark and its
close, and a transcript's own bracket used to move that cut.
This commit is contained in:
Jinwoo-H
2026-09-10 23:29:42 -04:00
parent 7b91da9552
commit 97e21051d2
2 changed files with 108 additions and 12 deletions
@@ -0,0 +1,75 @@
import { afterEach, expect, it } from 'vitest'
import {
SESSION_SEARCH_SNIPPET_MARK_CLOSE,
SESSION_SEARCH_SNIPPET_MARK_OPEN
} from './session-search-engine-types'
import {
addSyntheticSession,
openSessionSearchHarness,
type SessionSearchHarness
} from './session-search-engine-test-fixture'
// A snippet has to name which of a row's four columns matched, and the marks
// FTS5 wraps a match in are the only signal. Searching the marked text for the
// public `[[` reads a transcript's own brackets as a highlight — and transcripts
// are full of them, because a bash `[[ -f x ]]` and numpy's `[[1, 2]]` are
// exactly the sort of thing an agent session holds. Whether a column matched is
// the difference between two renderings of the same text instead.
let harness: SessionSearchHarness | null = null
afterEach(async () => {
await harness?.close()
harness = null
})
const BASH = 'run this: if [[ -f /home/me/.aws/credentials ]]; then cat it; fi'
const TOOL = 'zebrafish appears only in the tool output here'
it('shows the column that matched, not the one that happens to contain brackets', async () => {
harness = await openSessionSearchHarness('ss-snippet-marks')
// Session 1's match is in tool output while its user turn holds a bash test
// expression; session 2 is the same match with no brackets anywhere.
addSyntheticSession(harness.db, { id: 1, text: BASH, toolText: TOOL })
addSyntheticSession(harness.db, { id: 2, text: 'run this script please', toolText: TOOL })
const hits = harness.engine.search({ query: 'zebrafish' }).hits
expect(hits).toHaveLength(2)
for (const hit of hits) {
expect(hit.evidence?.snippet).toContain(
`${SESSION_SEARCH_SNIPPET_MARK_OPEN}zebrafish${SESSION_SEARCH_SNIPPET_MARK_CLOSE}`
)
expect(hit.evidence?.snippet).not.toContain('credentials')
}
})
it('falls back to any column for an identifier-only match, brackets or not', async () => {
// `zebra` reaches this row only through the identifier shadow column, which is
// what column -1 exists for. The user turn holds numpy output, so a bracket
// scan would have stopped at it and shown a column with no match in it.
harness = await openSessionSearchHarness('ss-snippet-marks-fallback')
addSyntheticSession(harness.db, {
id: 1,
text: 'numpy printed [[1, 2], [3, 4]] before the call',
toolText: 'zebra-fish-count = 4'
})
const [hit] = harness.engine.search({ query: 'zebra' }).hits
expect(hit?.evidence?.snippet).toContain(
`${SESSION_SEARCH_SNIPPET_MARK_OPEN}zebra${SESSION_SEARCH_SNIPPET_MARK_CLOSE}`
)
expect(hit?.evidence?.snippet).not.toContain('numpy')
})
it('leaves a transcript’s own brackets in the text it shows', async () => {
// The marks are rewritten from private-use code points at the very end, so a
// row that both matches and contains `[[` keeps its own characters.
harness = await openSessionSearchHarness('ss-snippet-marks-literal')
addSyntheticSession(harness.db, { id: 1, text: `zebrafish ${BASH}` })
const [hit] = harness.engine.search({ query: 'zebrafish' }).hits
expect(hit?.evidence?.snippet).toContain(
`${SESSION_SEARCH_SNIPPET_MARK_OPEN}zebrafish${SESSION_SEARCH_SNIPPET_MARK_CLOSE}`
)
expect(hit?.evidence?.snippet).toContain('[[ -f')
})
@@ -7,6 +7,15 @@ import { orExpression, type SessionSearchQueryPlan } from './session-search-quer
import { scopedExpression } from './session-search-retrieval'
import type { SessionSearchScope } from './session-search-engine-types'
// What FTS5 wraps a match in before this module rewrites it to the public
// marks. Private-use code points, and not `[[`, because two different jobs here
// have to tell a mark from content: choosing the column to show, and refusing
// to cut a snippet between an open mark and its close. Transcripts contain
// `[[` — a bash `[[ -f x ]]`, numpy's `[[1, 2]]` — and a mark the content can
// forge makes both of those decisions wrong on real text.
const MARK_OPEN = '\uE000'
const MARK_CLOSE = '\uE001'
const SNIPPET_TOKENS = 12
// Why a ceiling on top of the token count: a transcript chunk can be 8000
// characters with no separator in it, which FTS5 reports as one token, so
@@ -45,11 +54,16 @@ export function sessionSearchSnippet(
// second list here would be a guard with nothing left to guard, and the two
// would mask each other's mistakes.
const columns = [0, 1, 2, -1]
// Each column twice: once marked, once with empty marks. Whether a column
// matched is then the difference between two renderings of the same text,
// which content cannot forge — searching the marked one for a mark reads a
// transcript's own `[[` as a highlight and shows a column that matched
// nothing.
const select = columns
.map(
(column, index) =>
`snippet(messages_fts, ${column}, '${SESSION_SEARCH_SNIPPET_MARK_OPEN}', '${SESSION_SEARCH_SNIPPET_MARK_CLOSE}', '…', ${SNIPPET_TOKENS}) AS c${index}`
)
.flatMap((column, index) => [
`snippet(messages_fts, ${column}, '${MARK_OPEN}', '${MARK_CLOSE}', '…', ${SNIPPET_TOKENS}) AS c${index}`,
`snippet(messages_fts, ${column}, '', '', '…', ${SNIPPET_TOKENS}) AS p${index}`
])
.join(', ')
try {
// Why the subselect: a bound `rowid = ?` or `rowid IN (?)` next to MATCH is
@@ -75,14 +89,24 @@ export function sessionSearchSnippet(
// A snippet with nothing highlighted tells the user nothing; omit it.
const marked = columns
.map((_column, index) => row[`c${index}`])
.find((text) => text?.includes(SESSION_SEARCH_SNIPPET_MARK_OPEN))
return marked === undefined ? EMPTY_SNIPPET : truncateSnippet(marked)
.find((text, index) => text !== undefined && text !== row[`p${index}`])
return marked === undefined ? EMPTY_SNIPPET : publicMarks(truncateSnippet(marked))
} catch {
return EMPTY_SNIPPET
}
}
/** Cut on a code-point boundary, and never between `[[` and its `]]`. */
/** The internal marks, swapped for the ones a caller sees, once and at the end. */
function publicMarks(snippet: SessionSearchSnippet): SessionSearchSnippet {
return {
...snippet,
text: snippet.text
.replaceAll(MARK_OPEN, SESSION_SEARCH_SNIPPET_MARK_OPEN)
.replaceAll(MARK_CLOSE, SESSION_SEARCH_SNIPPET_MARK_CLOSE)
}
}
/** Cut on a code-point boundary, and never between a mark and its close. */
export function truncateSnippet(text: string): SessionSearchSnippet {
if (text.length <= SNIPPET_MAX_CHARS) {
return { text, truncated: false }
@@ -92,11 +116,8 @@ export function truncateSnippet(text: string): SessionSearchSnippet {
return { text, truncated: false }
}
const cut = points.slice(0, SNIPPET_MAX_CHARS).join('')
const opened = cut.lastIndexOf(SESSION_SEARCH_SNIPPET_MARK_OPEN)
const opened = cut.lastIndexOf(MARK_OPEN)
// An open mark with no close hands the renderer something it can never close.
const balanced =
opened !== -1 && !cut.includes(SESSION_SEARCH_SNIPPET_MARK_CLOSE, opened)
? cut.slice(0, opened)
: cut
const balanced = opened !== -1 && !cut.includes(MARK_CLOSE, opened) ? cut.slice(0, opened) : cut
return { text: balanced, truncated: true }
}