test(ai-vault-search): make each snippet mark mechanism answer for itself

Two mechanisms landed together and hid each other: choosing a column by
comparing a marked rendering against an unmarked one, and marking with
private-use code points instead of `[[`. Either alone fixed the bracket repro,
so neither had a mutation against it — the same masking the snippet's two
column guards had a round ago.

They do different jobs, so both stay and each gets the test that needs it. A
transcript holding a private-use code point of its own is what the comparison
is for; agent output carries Nerd Font glyphs from that block. A snippet past
the character ceiling with a bracket after its last real mark is what the
private-use marks are for, because the truncation has to find that mark by
searching the text.

The two one-line fixes get honest framing rather than a mutation neither can
have. A lone surrogate is not a token character, so the planner drops it either
way and the safe cut is hygiene. And no SQLite this stack runs refuses 1,100
bound ids — 32,766 has been the floor since 3.32 — so the batch is about
owning the ceiling here rather than rescuing a reachable failure.
This commit is contained in:
Jinwoo-H
2026-09-10 23:29:43 -04:00
parent 1eee4ba7fa
commit 33fbd00d25
3 changed files with 61 additions and 18 deletions
@@ -395,23 +395,24 @@ describe('the engine carries its own schema and puts it back', () => {
})
describe('a query the engine had to cut says so', () => {
it('cuts the query on a whole code point, never through a surrogate pair', async () => {
// A bare slice at 512 can land between the two halves of an astral
// character, and the lone half matches nothing and cannot be echoed back.
it('answers a query whose cap falls inside an astral character', async () => {
// The cut is on a whole code point rather than a code unit, so nothing
// downstream is handed half a surrogate pair. That is hygiene rather than a
// behaviour: the planner's tokenizer does not treat a lone surrogate as a
// token character, so it drops out of the terms either way. What this pins
// is that the boundary is answerable at all.
const { db, engine } = await open('ss-engine-surrogate-cap')
addSyntheticSession(db, { id: 1, text: 'needle' })
const query = `${'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 1)}😀 needle`
const result = engine.search({ query })
const kept = 'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 2)
addSyntheticSession(db, { id: 1, text: kept })
const result = engine.search({ query: `${kept} 😀 tail` })
expect(result.truncated.query).toBe(true)
// The emoji straddles the cap, so the cut has to fall before it.
expect(query.slice(0, SESSION_SEARCH_QUERY_MAX_LENGTH).at(-1)).toBe('\ud83d')
expect(result.hits).toEqual([])
expect(result.hits.map((hit) => hit.sessionId)).toEqual(['1'])
})
it('loads more candidate sessions than SQLite will bind in one statement', async () => {
it('loads a candidate set larger than one batch of bound ids', async () => {
// The id list is as long as the candidate limit and every id is a bound
// parameter, so one statement is a raised limit away from `too many SQL
// variables` on a host whose SQLite caps at 999.
// parameter. No SQLite this stack can run refuses 1,100 of them, so this
// pins that batching returns the same answer, not that it rescues one.
const { db, engine } = await open('ss-engine-id-batching', {
sessionCandidateLimit: 1200
})
@@ -14,8 +14,8 @@ import { SessionSearchTypoRepair } from './session-search-typo-repair'
// The operator-only walk: rows per page, and how far past a full candidate set
// it will read before giving up on finding more matches.
const RECENT_PAGE_ROWS = 512
// Ids per `loadSessions` statement, left well under the 999-parameter floor so
// the filter's own bound values fit beside them.
// Ids per `loadSessions` statement, with room to spare for the filter's own
// bound values beside them.
const SESSION_ID_BATCH = 500
const RECENT_SCAN_FACTOR = 20
@@ -165,10 +165,15 @@ export class SessionSearchRetrieval {
/**
* Read in batches, because the id list is as long as the candidate limit and
* every id is a bound parameter. SQLite's default `SQLITE_MAX_VARIABLE_NUMBER`
* is 999 on builds older than 3.32, and a caller may raise the candidate
* limit — the tuning doc says it may — so a single statement is one settings
* change away from `too many SQL variables` on somebody's host.
* every id is a bound parameter, so a single statement scales with a knob the
* tuning doc invites a host to raise.
*
* Not a fix for a reachable failure, and worth saying so: SQLite has bound
* `SQLITE_MAX_VARIABLE_NUMBER` at 32,766 since 3.32, every runtime this stack
* supports is past that, and the measured limit on this one is higher still.
* A candidate limit that large is not a configuration anyone would choose.
* The batch is here so the ceiling belongs to this file rather than to
* whichever SQLite the process happened to link.
*/
loadSessions(ids: readonly number[], scope: RetrievalScope): SessionRow[] {
const rows: SessionRow[] = []
@@ -73,3 +73,40 @@ it('leaves a transcript’s own brackets in the text it shows', async () => {
)
expect(hit?.evidence?.snippet).toContain('[[ -f')
})
it('picks by comparison, so a private-use code point in content cannot pose as a mark', async () => {
// The marks are private-use code points, and a transcript may hold one:
// agent output carries Nerd Font glyphs, which live in the same block. So the
// column is chosen by comparing a marked rendering against an unmarked one,
// not by looking for a mark in the text.
harness = await openSessionSearchHarness('ss-snippet-marks-private-use')
addSyntheticSession(harness.db, {
id: 1,
text: 'the \uE000 glyph a font printed here',
toolText: TOOL
})
const [hit] = harness.engine.search({ query: 'zebrafish' }).hits
expect(hit?.evidence?.snippet).toContain('zebrafish')
expect(hit?.evidence?.snippet).not.toContain('glyph')
})
it('truncates on the last real mark, not on a bracket the transcript wrote', async () => {
// Over the character ceiling the snippet is cut, and it must not cut between
// an open mark and its close. Finding that open mark by searching for `[[`
// stops at the transcript's own bracket instead and throws away everything
// after it.
harness = await openSessionSearchHarness('ss-snippet-marks-truncation')
const long = (letter: string): string =>
Array.from({ length: 5 }, () => `${letter.repeat(55)}/tail`).join(' ')
addSyntheticSession(harness.db, {
id: 1,
text: `zebrafish ${long('p')} [[ ${long('q')}`
})
const snippet = harness.engine.search({ query: 'zebrafish' }).hits[0]?.evidence?.snippet ?? ''
expect(snippet).toContain('[[zebrafish]]')
// The cut is the character ceiling, so the text after the transcript's own
// bracket survives up to it.
expect(snippet).toContain('qqqqq')
})