mirror of
https://github.com/stablyai/orca.git
synced 2026-10-03 00:02:19 +00:00
test(ai-vault-search): make each snippet mark mechanism answer for itself
Two mechanisms landed together and hid each other: choosing a column by comparing a marked rendering against an unmarked one, and marking with private-use code points instead of `[[`. Either alone fixed the bracket repro, so neither had a mutation against it — the same masking the snippet's two column guards had a round ago. They do different jobs, so both stay and each gets the test that needs it. A transcript holding a private-use code point of its own is what the comparison is for; agent output carries Nerd Font glyphs from that block. A snippet past the character ceiling with a bracket after its last real mark is what the private-use marks are for, because the truncation has to find that mark by searching the text. The two one-line fixes get honest framing rather than a mutation neither can have. A lone surrogate is not a token character, so the planner drops it either way and the safe cut is hygiene. And no SQLite this stack runs refuses 1,100 bound ids — 32,766 has been the floor since 3.32 — so the batch is about owning the ceiling here rather than rescuing a reachable failure.
This commit is contained in:
@@ -395,23 +395,24 @@ describe('the engine carries its own schema and puts it back', () => {
|
||||
})
|
||||
|
||||
describe('a query the engine had to cut says so', () => {
|
||||
it('cuts the query on a whole code point, never through a surrogate pair', async () => {
|
||||
// A bare slice at 512 can land between the two halves of an astral
|
||||
// character, and the lone half matches nothing and cannot be echoed back.
|
||||
it('answers a query whose cap falls inside an astral character', async () => {
|
||||
// The cut is on a whole code point rather than a code unit, so nothing
|
||||
// downstream is handed half a surrogate pair. That is hygiene rather than a
|
||||
// behaviour: the planner's tokenizer does not treat a lone surrogate as a
|
||||
// token character, so it drops out of the terms either way. What this pins
|
||||
// is that the boundary is answerable at all.
|
||||
const { db, engine } = await open('ss-engine-surrogate-cap')
|
||||
addSyntheticSession(db, { id: 1, text: 'needle' })
|
||||
const query = `${'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 1)}😀 needle`
|
||||
const result = engine.search({ query })
|
||||
const kept = 'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 2)
|
||||
addSyntheticSession(db, { id: 1, text: kept })
|
||||
const result = engine.search({ query: `${kept} 😀 tail` })
|
||||
expect(result.truncated.query).toBe(true)
|
||||
// The emoji straddles the cap, so the cut has to fall before it.
|
||||
expect(query.slice(0, SESSION_SEARCH_QUERY_MAX_LENGTH).at(-1)).toBe('\ud83d')
|
||||
expect(result.hits).toEqual([])
|
||||
expect(result.hits.map((hit) => hit.sessionId)).toEqual(['1'])
|
||||
})
|
||||
|
||||
it('loads more candidate sessions than SQLite will bind in one statement', async () => {
|
||||
it('loads a candidate set larger than one batch of bound ids', async () => {
|
||||
// The id list is as long as the candidate limit and every id is a bound
|
||||
// parameter, so one statement is a raised limit away from `too many SQL
|
||||
// variables` on a host whose SQLite caps at 999.
|
||||
// parameter. No SQLite this stack can run refuses 1,100 of them, so this
|
||||
// pins that batching returns the same answer, not that it rescues one.
|
||||
const { db, engine } = await open('ss-engine-id-batching', {
|
||||
sessionCandidateLimit: 1200
|
||||
})
|
||||
|
||||
@@ -14,8 +14,8 @@ import { SessionSearchTypoRepair } from './session-search-typo-repair'
|
||||
// The operator-only walk: rows per page, and how far past a full candidate set
|
||||
// it will read before giving up on finding more matches.
|
||||
const RECENT_PAGE_ROWS = 512
|
||||
// Ids per `loadSessions` statement, left well under the 999-parameter floor so
|
||||
// the filter's own bound values fit beside them.
|
||||
// Ids per `loadSessions` statement, with room to spare for the filter's own
|
||||
// bound values beside them.
|
||||
const SESSION_ID_BATCH = 500
|
||||
const RECENT_SCAN_FACTOR = 20
|
||||
|
||||
@@ -165,10 +165,15 @@ export class SessionSearchRetrieval {
|
||||
|
||||
/**
|
||||
* Read in batches, because the id list is as long as the candidate limit and
|
||||
* every id is a bound parameter. SQLite's default `SQLITE_MAX_VARIABLE_NUMBER`
|
||||
* is 999 on builds older than 3.32, and a caller may raise the candidate
|
||||
* limit — the tuning doc says it may — so a single statement is one settings
|
||||
* change away from `too many SQL variables` on somebody's host.
|
||||
* every id is a bound parameter, so a single statement scales with a knob the
|
||||
* tuning doc invites a host to raise.
|
||||
*
|
||||
* Not a fix for a reachable failure, and worth saying so: SQLite has bound
|
||||
* `SQLITE_MAX_VARIABLE_NUMBER` at 32,766 since 3.32, every runtime this stack
|
||||
* supports is past that, and the measured limit on this one is higher still.
|
||||
* A candidate limit that large is not a configuration anyone would choose.
|
||||
* The batch is here so the ceiling belongs to this file rather than to
|
||||
* whichever SQLite the process happened to link.
|
||||
*/
|
||||
loadSessions(ids: readonly number[], scope: RetrievalScope): SessionRow[] {
|
||||
const rows: SessionRow[] = []
|
||||
|
||||
@@ -73,3 +73,40 @@ it('leaves a transcript’s own brackets in the text it shows', async () => {
|
||||
)
|
||||
expect(hit?.evidence?.snippet).toContain('[[ -f')
|
||||
})
|
||||
|
||||
it('picks by comparison, so a private-use code point in content cannot pose as a mark', async () => {
|
||||
// The marks are private-use code points, and a transcript may hold one:
|
||||
// agent output carries Nerd Font glyphs, which live in the same block. So the
|
||||
// column is chosen by comparing a marked rendering against an unmarked one,
|
||||
// not by looking for a mark in the text.
|
||||
harness = await openSessionSearchHarness('ss-snippet-marks-private-use')
|
||||
addSyntheticSession(harness.db, {
|
||||
id: 1,
|
||||
text: 'the \uE000 glyph a font printed here',
|
||||
toolText: TOOL
|
||||
})
|
||||
|
||||
const [hit] = harness.engine.search({ query: 'zebrafish' }).hits
|
||||
expect(hit?.evidence?.snippet).toContain('zebrafish')
|
||||
expect(hit?.evidence?.snippet).not.toContain('glyph')
|
||||
})
|
||||
|
||||
it('truncates on the last real mark, not on a bracket the transcript wrote', async () => {
|
||||
// Over the character ceiling the snippet is cut, and it must not cut between
|
||||
// an open mark and its close. Finding that open mark by searching for `[[`
|
||||
// stops at the transcript's own bracket instead and throws away everything
|
||||
// after it.
|
||||
harness = await openSessionSearchHarness('ss-snippet-marks-truncation')
|
||||
const long = (letter: string): string =>
|
||||
Array.from({ length: 5 }, () => `${letter.repeat(55)}/tail`).join(' ')
|
||||
addSyntheticSession(harness.db, {
|
||||
id: 1,
|
||||
text: `zebrafish ${long('p')} [[ ${long('q')}`
|
||||
})
|
||||
|
||||
const snippet = harness.engine.search({ query: 'zebrafish' }).hits[0]?.evidence?.snippet ?? ''
|
||||
expect(snippet).toContain('[[zebrafish]]')
|
||||
// The cut is the character ceiling, so the text after the transcript's own
|
||||
// bracket survives up to it.
|
||||
expect(snippet).toContain('qqqqq')
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user