mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 08:02:43 +00:00
test(ai-vault-search): pin the two engine seams nothing was holding
The warm wiring and the query-length cap were both written and neither was observable. A spy pins that a search is what warms the store, and the cap is pinned through the one input the planner's own term limit does not already bound: a single enormous token, where the cut is what decides whether the term matches the indexed one at all.
This commit is contained in:
@@ -1,4 +1,5 @@
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { SESSION_SEARCH_QUERY_MAX_LENGTH } from './session-search-engine-types'
|
||||
import type { SessionSearchRequest, SessionSearchResponse } from './session-search-engine-types'
|
||||
import {
|
||||
addSyntheticSession,
|
||||
@@ -254,6 +255,33 @@ describe('source presence comes from the files table, never a stat', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('the engine is the only thing that warms the index', () => {
|
||||
it('warms on the first search and leans on the store to memoize the rest', async () => {
|
||||
const { db, store, engine } = await open('ss-engine-warm')
|
||||
addSyntheticSession(db, { id: 1 })
|
||||
const warm = vi.spyOn(store, 'warm')
|
||||
engine.search({ query: 'needle' })
|
||||
engine.search({ query: 'needle' })
|
||||
// PR 2 left the call site to whoever knows which pages a read touches. The
|
||||
// store returns one memoized promise, so asking twice costs one warm-up.
|
||||
expect(warm).toHaveBeenCalledTimes(2)
|
||||
await store.warm()
|
||||
})
|
||||
})
|
||||
|
||||
describe('a query longer than the engine will plan is cut, not refused', () => {
|
||||
it('cuts one enormous token down to the cap before FTS5 ever sees it', async () => {
|
||||
const { db, engine } = await open('ss-engine-long-query')
|
||||
// The planner already caps how many terms it will plan, so a long query of
|
||||
// ordinary words is bounded without this. What is not bounded is a single
|
||||
// token: one 100 kB word is one term, and FTS5 would carry the whole thing
|
||||
// into the MATCH expression. The cut is observable because the indexed
|
||||
// token is exactly the capped length.
|
||||
addSyntheticSession(db, { id: 1, text: 'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH) })
|
||||
expect(ids(engine.search({ query: 'x'.repeat(4000) }))).toEqual(['1'])
|
||||
})
|
||||
})
|
||||
|
||||
describe('unicode terms survive the round trip', () => {
|
||||
it.each(['café', 'C', 'R', 'x', '修復', '안녕하세요'])('searches %s', async (text) => {
|
||||
const { db, engine } = await open('ss-engine-unicode')
|
||||
|
||||
Reference in New Issue
Block a user