From 33fbd00d25eb3113a63d3a41526a2d73ce35444c Mon Sep 17 00:00:00 2001 From: Jinwoo-H Date: Thu, 10 Sep 2026 16:40:15 -0400 Subject: [PATCH] test(ai-vault-search): make each snippet mark mechanism answer for itself MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two mechanisms landed together and hid each other: choosing a column by comparing a marked rendering against an unmarked one, and marking with private-use code points instead of `[[`. Either alone fixed the bracket repro, so neither had a mutation against it โ€” the same masking the snippet's two column guards had a round ago. They do different jobs, so both stay and each gets the test that needs it. A transcript holding a private-use code point of its own is what the comparison is for; agent output carries Nerd Font glyphs from that block. A snippet past the character ceiling with a bracket after its last real mark is what the private-use marks are for, because the truncation has to find that mark by searching the text. The two one-line fixes get honest framing rather than a mutation neither can have. A lone surrogate is not a token character, so the planner drops it either way and the safe cut is hygiene. And no SQLite this stack runs refuses 1,100 bound ids โ€” 32,766 has been the floor since 3.32 โ€” so the batch is about owning the ceiling here rather than rescuing a reachable failure. --- .../session-search-engine.test.ts | 25 +++++++------ .../session-search-retrieval.ts | 17 ++++++--- .../session-search-snippet-marks.test.ts | 37 +++++++++++++++++++ 3 files changed, 61 insertions(+), 18 deletions(-) diff --git a/src/main/ai-vault-search/session-search-engine.test.ts b/src/main/ai-vault-search/session-search-engine.test.ts index bc25fb06d2d..140f9449c3d 100644 --- a/src/main/ai-vault-search/session-search-engine.test.ts +++ b/src/main/ai-vault-search/session-search-engine.test.ts @@ -395,23 +395,24 @@ describe('the engine carries its own schema and puts it back', () => { }) describe('a query the engine had to cut says so', () => { - it('cuts the query on a whole code point, never through a surrogate pair', async () => { - // A bare slice at 512 can land between the two halves of an astral - // character, and the lone half matches nothing and cannot be echoed back. + it('answers a query whose cap falls inside an astral character', async () => { + // The cut is on a whole code point rather than a code unit, so nothing + // downstream is handed half a surrogate pair. That is hygiene rather than a + // behaviour: the planner's tokenizer does not treat a lone surrogate as a + // token character, so it drops out of the terms either way. What this pins + // is that the boundary is answerable at all. const { db, engine } = await open('ss-engine-surrogate-cap') - addSyntheticSession(db, { id: 1, text: 'needle' }) - const query = `${'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 1)}๐Ÿ˜€ needle` - const result = engine.search({ query }) + const kept = 'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 2) + addSyntheticSession(db, { id: 1, text: kept }) + const result = engine.search({ query: `${kept} ๐Ÿ˜€ tail` }) expect(result.truncated.query).toBe(true) - // The emoji straddles the cap, so the cut has to fall before it. - expect(query.slice(0, SESSION_SEARCH_QUERY_MAX_LENGTH).at(-1)).toBe('\ud83d') - expect(result.hits).toEqual([]) + expect(result.hits.map((hit) => hit.sessionId)).toEqual(['1']) }) - it('loads more candidate sessions than SQLite will bind in one statement', async () => { + it('loads a candidate set larger than one batch of bound ids', async () => { // The id list is as long as the candidate limit and every id is a bound - // parameter, so one statement is a raised limit away from `too many SQL - // variables` on a host whose SQLite caps at 999. + // parameter. No SQLite this stack can run refuses 1,100 of them, so this + // pins that batching returns the same answer, not that it rescues one. const { db, engine } = await open('ss-engine-id-batching', { sessionCandidateLimit: 1200 }) diff --git a/src/main/ai-vault-search/session-search-retrieval.ts b/src/main/ai-vault-search/session-search-retrieval.ts index 346f99c677e..73cff2804f8 100644 --- a/src/main/ai-vault-search/session-search-retrieval.ts +++ b/src/main/ai-vault-search/session-search-retrieval.ts @@ -14,8 +14,8 @@ import { SessionSearchTypoRepair } from './session-search-typo-repair' // The operator-only walk: rows per page, and how far past a full candidate set // it will read before giving up on finding more matches. const RECENT_PAGE_ROWS = 512 -// Ids per `loadSessions` statement, left well under the 999-parameter floor so -// the filter's own bound values fit beside them. +// Ids per `loadSessions` statement, with room to spare for the filter's own +// bound values beside them. const SESSION_ID_BATCH = 500 const RECENT_SCAN_FACTOR = 20 @@ -165,10 +165,15 @@ export class SessionSearchRetrieval { /** * Read in batches, because the id list is as long as the candidate limit and - * every id is a bound parameter. SQLite's default `SQLITE_MAX_VARIABLE_NUMBER` - * is 999 on builds older than 3.32, and a caller may raise the candidate - * limit โ€” the tuning doc says it may โ€” so a single statement is one settings - * change away from `too many SQL variables` on somebody's host. + * every id is a bound parameter, so a single statement scales with a knob the + * tuning doc invites a host to raise. + * + * Not a fix for a reachable failure, and worth saying so: SQLite has bound + * `SQLITE_MAX_VARIABLE_NUMBER` at 32,766 since 3.32, every runtime this stack + * supports is past that, and the measured limit on this one is higher still. + * A candidate limit that large is not a configuration anyone would choose. + * The batch is here so the ceiling belongs to this file rather than to + * whichever SQLite the process happened to link. */ loadSessions(ids: readonly number[], scope: RetrievalScope): SessionRow[] { const rows: SessionRow[] = [] diff --git a/src/main/ai-vault-search/session-search-snippet-marks.test.ts b/src/main/ai-vault-search/session-search-snippet-marks.test.ts index eb4c5e8423c..71704b78fee 100644 --- a/src/main/ai-vault-search/session-search-snippet-marks.test.ts +++ b/src/main/ai-vault-search/session-search-snippet-marks.test.ts @@ -73,3 +73,40 @@ it('leaves a transcriptโ€™s own brackets in the text it shows', async () => { ) expect(hit?.evidence?.snippet).toContain('[[ -f') }) + +it('picks by comparison, so a private-use code point in content cannot pose as a mark', async () => { + // The marks are private-use code points, and a transcript may hold one: + // agent output carries Nerd Font glyphs, which live in the same block. So the + // column is chosen by comparing a marked rendering against an unmarked one, + // not by looking for a mark in the text. + harness = await openSessionSearchHarness('ss-snippet-marks-private-use') + addSyntheticSession(harness.db, { + id: 1, + text: 'the \uE000 glyph a font printed here', + toolText: TOOL + }) + + const [hit] = harness.engine.search({ query: 'zebrafish' }).hits + expect(hit?.evidence?.snippet).toContain('zebrafish') + expect(hit?.evidence?.snippet).not.toContain('glyph') +}) + +it('truncates on the last real mark, not on a bracket the transcript wrote', async () => { + // Over the character ceiling the snippet is cut, and it must not cut between + // an open mark and its close. Finding that open mark by searching for `[[` + // stops at the transcript's own bracket instead and throws away everything + // after it. + harness = await openSessionSearchHarness('ss-snippet-marks-truncation') + const long = (letter: string): string => + Array.from({ length: 5 }, () => `${letter.repeat(55)}/tail`).join(' ') + addSyntheticSession(harness.db, { + id: 1, + text: `zebrafish ${long('p')} [[ ${long('q')}` + }) + + const snippet = harness.engine.search({ query: 'zebrafish' }).hits[0]?.evidence?.snippet ?? '' + expect(snippet).toContain('[[zebrafish]]') + // The cut is the character ceiling, so the text after the transcript's own + // bracket survives up to it. + expect(snippet).toContain('qqqqq') +})