mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 00:02:41 +00:00
refactor(ai-vault-search): answer the conversation scope with a column filter
PR 2 deleted `conversation_fts` on the strength of this PR's shoot-out, so the
scope is a column filter over the one FTS table now. `ftsTableFor` is gone; a
scope is a pair of `scopedExpression` and `scopedWeights`, and the table name
no longer travels through the engine, the snippet builder or a hit.
The filter is parenthesised, and that is the whole of it: `{cols}: (a AND b)`
binds both terms, while `{cols}: a AND b` binds only the first and searches
tool output for the rest. A test drives an AND whose second term lives only in
tool output through both scopes.
The snippet keeps one guard, not two. Its column list and its expression were
each hiding the other's mistakes — a tool-only row was unreachable through
either — so the list is the same four columns for every scope and the scoped
expression is what makes a conversation snippet impossible to draw out of tool
output. Dropping it now leaks that row, which a test catches.
One behaviour the deleted table did not have, pinned rather than wished away:
bm25 normalises by the whole row's length and has no per-column length, so two
rows with identical prose score differently when one also holds tool output.
The rowid set is unchanged; the order within it can move.
Re-measured on the shipping schema. The conversation scope is 1.2-1.4x faster
than `all` at every rung, and the index is 57 MB rather than about 150 MB at
93% tool output, because a tool row is now capped at 3,072 characters.
This commit is contained in:
@@ -1,270 +0,0 @@
|
||||
import { rm, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import {
|
||||
createSessionParseStats,
|
||||
parseAgentSessionFileCached,
|
||||
resetSessionParseCacheForTests
|
||||
} from '../../src/main/ai-vault/session-scanner-parse-cache'
|
||||
import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers'
|
||||
import {
|
||||
andExpression,
|
||||
phraseExpression,
|
||||
planSessionSearchQuery
|
||||
} from '../../src/main/ai-vault-search/session-search-query-planner'
|
||||
import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer'
|
||||
import { openSessionSearchDatabase } from '../../src/main/ai-vault-search/session-search-schema'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures'
|
||||
import type SyncDatabase from '../../src/main/sqlite/sync-database'
|
||||
import { writeToolHeavyCorpus, type ToolHeavyCorpus } from './session-search-tool-heavy-corpus'
|
||||
|
||||
// Open decision 3: `conversation_fts` holds a copy of the two prose columns and
|
||||
// costs about a quarter of the index. A column filter over `messages_fts`
|
||||
// returns the same rows, so the only question is what it costs to read them out
|
||||
// of a table that also holds every byte of tool output. This is that
|
||||
// measurement, on a corpus sized and shaped like a real transcript tree (see
|
||||
// session-search-tool-heavy-corpus).
|
||||
|
||||
const WARMUP = 5
|
||||
|
||||
/** The queries the two arms answer; every one is a conversation-scope shape. */
|
||||
const QUERIES = [
|
||||
'terminal reattach',
|
||||
'stale snapshot',
|
||||
'daemon cursor',
|
||||
'worktree index',
|
||||
'publish transaction',
|
||||
'relay daemon',
|
||||
'session cursor',
|
||||
'because stale',
|
||||
'terminal worktree',
|
||||
'index snapshot',
|
||||
'reattach cursor',
|
||||
'transaction relay',
|
||||
'snapshot session',
|
||||
'daemon publish',
|
||||
'worktree terminal',
|
||||
'cursor index',
|
||||
'stale relay',
|
||||
'session transaction',
|
||||
'publish snapshot',
|
||||
'reattach daemon'
|
||||
]
|
||||
|
||||
async function indexCorpus(
|
||||
corpus: ToolHeavyCorpus
|
||||
): Promise<{ db: SyncDatabase; release: () => void }> {
|
||||
resetSessionParseCacheForTests()
|
||||
const indexPath = join(corpus.root, 'index.sqlite')
|
||||
const store = new SessionSearchStore(indexPath, (error) => {
|
||||
throw error
|
||||
})
|
||||
const unregister = registerSessionSearchIndexConsumer(store)
|
||||
const stats = createSessionParseStats()
|
||||
for (const path of corpus.files) {
|
||||
await parseAgentSessionFileCached(
|
||||
await sessionCandidate('claude', path),
|
||||
process.platform,
|
||||
stats
|
||||
)
|
||||
}
|
||||
const db = openSessionSearchDatabase(indexPath)
|
||||
return {
|
||||
db,
|
||||
release: () => {
|
||||
unregister()
|
||||
resetTranscriptConsumersForTests()
|
||||
resetSessionParseCacheForTests()
|
||||
db.close()
|
||||
store.close()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The engine's own retrieval shape, so the two arms differ in nothing but the
|
||||
* table they read and the expression that names the columns.
|
||||
*/
|
||||
function matchSql(table: string, weights: string): string {
|
||||
return `WITH matched AS MATERIALIZED (
|
||||
SELECT ${table}.rowid AS rowid, -bm25(${table}, ${weights}) AS score,
|
||||
m.session_row_id, m.role, m.ts, s.updated_at
|
||||
FROM ${table} JOIN messages m ON m.id = ${table}.rowid
|
||||
JOIN sessions s ON s.id = m.session_row_id WHERE ${table} MATCH ?)
|
||||
SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched
|
||||
GROUP BY session_row_id ORDER BY score DESC LIMIT 600`
|
||||
}
|
||||
|
||||
const ARMS = {
|
||||
conversation: { table: 'conversation_fts', weights: '3.0, 2.0', filtered: false },
|
||||
columnFiltered: { table: 'messages_fts', weights: '3.0, 2.0, 0.0, 0.0', filtered: true }
|
||||
} as const
|
||||
|
||||
/** `{user_text assistant_text} : (expr)` is what makes the wide table answer the narrow one. */
|
||||
function expressionFor(arm: keyof typeof ARMS, expression: string): string {
|
||||
return ARMS[arm].filtered ? `{user_text assistant_text} : (${expression})` : expression
|
||||
}
|
||||
|
||||
type Timing = { p50: number; p95: number }
|
||||
|
||||
function timing(samples: readonly number[]): Timing {
|
||||
const sorted = [...samples].sort((left, right) => left - right)
|
||||
const at = (fraction: number): number =>
|
||||
Math.round(
|
||||
(sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * fraction))] ?? 0) * 100
|
||||
) / 100
|
||||
return { p50: at(0.5), p95: at(0.95) }
|
||||
}
|
||||
|
||||
type Route = 'phrase' | 'and'
|
||||
|
||||
function expressions(route: Route): string[] {
|
||||
return QUERIES.map((query) => {
|
||||
const plan = planSessionSearchQuery(query)
|
||||
return route === 'phrase' ? phraseExpression(plan.body) : andExpression(plan.body)
|
||||
})
|
||||
}
|
||||
|
||||
/** Every arm's rowid set for one expression, so "identical" is checked, not assumed. */
|
||||
function rowids(db: SyncDatabase, arm: keyof typeof ARMS, expression: string): number[] {
|
||||
const { table } = ARMS[arm]
|
||||
return (
|
||||
db
|
||||
.prepare(
|
||||
`SELECT ${table}.rowid AS rowid FROM ${table}
|
||||
JOIN messages m ON m.id = ${table}.rowid
|
||||
JOIN sessions s ON s.id = m.session_row_id
|
||||
WHERE ${table} MATCH ? ORDER BY ${table}.rowid`
|
||||
)
|
||||
.all(expressionFor(arm, expression)) as { rowid: number }[]
|
||||
).map((row) => row.rowid)
|
||||
}
|
||||
|
||||
/** Rows the wide table matches before the column filter cuts them. */
|
||||
function unfilteredRowCount(db: SyncDatabase, expression: string): number {
|
||||
return Number(
|
||||
(
|
||||
db
|
||||
.prepare('SELECT count(*) AS c FROM messages_fts WHERE messages_fts MATCH ?')
|
||||
.get(expression) as { c: number }
|
||||
).c
|
||||
)
|
||||
}
|
||||
|
||||
function measure(db: SyncDatabase, route: Route): Record<string, unknown> {
|
||||
const built = expressions(route)
|
||||
const report: Record<string, unknown> = {}
|
||||
const statements = Object.fromEntries(
|
||||
Object.entries(ARMS).map(([name, arm]) => [name, db.prepare(matchSql(arm.table, arm.weights))])
|
||||
)
|
||||
// Interleaved arm by arm: run back to back, the first one pays for every page
|
||||
// the OS cache had not seen and the ordering moves p95 more than the arm does.
|
||||
const samples: Record<string, number[]> = { conversation: [], columnFiltered: [] }
|
||||
let matchedRows = 0
|
||||
for (let run = 0; run < WARMUP; run++) {
|
||||
for (const arm of Object.keys(ARMS) as (keyof typeof ARMS)[]) {
|
||||
for (const expression of built) {
|
||||
statements[arm]!.all(expressionFor(arm, expression))
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const expression of built) {
|
||||
for (const arm of Object.keys(ARMS) as (keyof typeof ARMS)[]) {
|
||||
const started = performance.now()
|
||||
const rows = statements[arm]!.all(expressionFor(arm, expression))
|
||||
samples[arm]!.push(performance.now() - started)
|
||||
if (arm === 'conversation') {
|
||||
matchedRows += rows.length
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const arm of Object.keys(ARMS) as (keyof typeof ARMS)[]) {
|
||||
report[arm] = timing(samples[arm]!)
|
||||
}
|
||||
const mismatched = built.filter(
|
||||
(expression) =>
|
||||
rowids(db, 'conversation', expression).join() !==
|
||||
rowids(db, 'columnFiltered', expression).join()
|
||||
)
|
||||
return {
|
||||
...report,
|
||||
queries: built.length,
|
||||
sessionsMatched: matchedRows,
|
||||
// What the column filter has to read past: the same expression without it.
|
||||
unfilteredRows: built.reduce(
|
||||
(total, expression) => total + unfilteredRowCount(db, expression),
|
||||
0
|
||||
),
|
||||
filteredRows: built.reduce(
|
||||
(total, expression) => total + rowids(db, 'columnFiltered', expression).length,
|
||||
0
|
||||
),
|
||||
identicalRowidSets: mismatched.length === 0,
|
||||
...(mismatched.length > 0 ? { mismatched } : {})
|
||||
}
|
||||
}
|
||||
|
||||
/** Bytes each FTS table occupies, so the trade has both halves. */
|
||||
function tableBytes(db: SyncDatabase): Record<string, number> {
|
||||
const sum = (prefix: string): number =>
|
||||
Number(
|
||||
(
|
||||
db
|
||||
.prepare('SELECT COALESCE(SUM(pgsize),0) AS bytes FROM dbstat WHERE name LIKE ?')
|
||||
.get(`${prefix}%`) as { bytes: number }
|
||||
).bytes
|
||||
)
|
||||
const total = Number(
|
||||
(db.prepare('SELECT COALESCE(SUM(pgsize),0) AS bytes FROM dbstat').get() as { bytes: number })
|
||||
.bytes
|
||||
)
|
||||
const conversationFts = sum('conversation_fts')
|
||||
return {
|
||||
total,
|
||||
messagesFts: sum('messages_fts'),
|
||||
conversationFts,
|
||||
// The other half of the trade: what deleting the table gives back.
|
||||
conversationFtsShare: Math.round((conversationFts / total) * 1000) / 1000
|
||||
}
|
||||
}
|
||||
|
||||
const targetMb = Number(process.env.CORPUS_MB ?? 100)
|
||||
const toolShare = Number(process.env.TOOL_SHARE ?? 0.9)
|
||||
const corpus = await writeToolHeavyCorpus({
|
||||
targetBytes: targetMb * 1024 * 1024,
|
||||
toolShare
|
||||
})
|
||||
let report: string
|
||||
const indexed = await indexCorpus(corpus)
|
||||
try {
|
||||
let sizes: Record<string, number> | { unavailable: string } = { unavailable: 'no dbstat' }
|
||||
try {
|
||||
sizes = tableBytes(indexed.db)
|
||||
} catch {
|
||||
// dbstat is a compile-time option; the latency numbers stand without it.
|
||||
}
|
||||
report = JSON.stringify(
|
||||
{
|
||||
corpus: {
|
||||
sessions: corpus.files.length,
|
||||
transcriptMb: Math.round((corpus.transcriptBytes / 1024 / 1024) * 100) / 100,
|
||||
toolShareOfMessageText:
|
||||
Math.round((corpus.toolBytes / (corpus.toolBytes + corpus.proseBytes)) * 1000) / 1000
|
||||
},
|
||||
indexBytes: sizes,
|
||||
phrase: measure(indexed.db, 'phrase'),
|
||||
and: measure(indexed.db, 'and')
|
||||
},
|
||||
null,
|
||||
2
|
||||
)
|
||||
} finally {
|
||||
indexed.release()
|
||||
await rm(corpus.root, { recursive: true, force: true })
|
||||
}
|
||||
|
||||
const out = process.env.BENCH_OUT
|
||||
if (out) {
|
||||
await writeFile(out, `${report}\n`)
|
||||
}
|
||||
console.log(report)
|
||||
@@ -12,7 +12,6 @@ import type {
|
||||
SessionSearchScope
|
||||
} from '../../src/main/ai-vault-search/session-search-engine-types'
|
||||
import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer'
|
||||
import { openSessionSearchDatabase } from '../../src/main/ai-vault-search/session-search-schema'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import type SyncDatabase from '../../src/main/sqlite/sync-database'
|
||||
import {
|
||||
@@ -64,8 +63,7 @@ async function indexCorpus(
|
||||
): Promise<{ corpus: SyntheticCorpus; db: SyncDatabase; release: () => void }> {
|
||||
resetSessionParseCacheForTests()
|
||||
const corpus = await writeSyntheticTranscriptCorpus(options)
|
||||
const indexPath = join(corpus.root, 'index.sqlite')
|
||||
const store = new SessionSearchStore(indexPath, (error) => {
|
||||
const store = new SessionSearchStore(join(corpus.root, 'index.sqlite'), (error) => {
|
||||
throw error
|
||||
})
|
||||
const unregister = registerSessionSearchIndexConsumer(store)
|
||||
@@ -77,17 +75,15 @@ async function indexCorpus(
|
||||
stats
|
||||
)
|
||||
}
|
||||
// The reader's own handle: the store keeps its connection private, and every
|
||||
// read here is a single statement, so a second one pins no WAL snapshot.
|
||||
const db = openSessionSearchDatabase(indexPath)
|
||||
return {
|
||||
corpus,
|
||||
db,
|
||||
// The handle a composed reader gets. Every read here is one synchronous
|
||||
// statement, which is the contract that comes with it.
|
||||
db: store.connection,
|
||||
release: () => {
|
||||
unregister()
|
||||
resetTranscriptConsumersForTests()
|
||||
resetSessionParseCacheForTests()
|
||||
db.close()
|
||||
store.close()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,205 @@
|
||||
import { rm, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import {
|
||||
createSessionParseStats,
|
||||
parseAgentSessionFileCached,
|
||||
resetSessionParseCacheForTests
|
||||
} from '../../src/main/ai-vault/session-scanner-parse-cache'
|
||||
import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers'
|
||||
import { SessionSearchEngine } from '../../src/main/ai-vault-search/session-search-engine'
|
||||
import type {
|
||||
SessionSearchRequest,
|
||||
SessionSearchScope
|
||||
} from '../../src/main/ai-vault-search/session-search-engine-types'
|
||||
import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures'
|
||||
import type SyncDatabase from '../../src/main/sqlite/sync-database'
|
||||
import { writeToolHeavyCorpus, type ToolHeavyCorpus } from './session-search-tool-heavy-corpus'
|
||||
|
||||
// What each scope costs on an index the size of a real transcript tree.
|
||||
//
|
||||
// The 10.5 MB corpus in `session-search-query-benchmark.ts` sizes the route
|
||||
// ladder; this one sizes the corpus. `conversation` is a column filter over the
|
||||
// one FTS table rather than a second table of its own, and the whole cost of
|
||||
// that decision is how much of `messages_fts` a conversation query has to read
|
||||
// past — which is set by how much of a transcript is tool output.
|
||||
//
|
||||
// Synthetic, always: this must never be pointed at a real transcript.
|
||||
|
||||
const WARMUP = 5
|
||||
|
||||
/** Conversation-shaped queries; every term is one the prose actually uses. */
|
||||
const QUERIES = [
|
||||
'terminal reattach',
|
||||
'stale snapshot',
|
||||
'daemon cursor',
|
||||
'worktree index',
|
||||
'publish transaction',
|
||||
'relay daemon',
|
||||
'session cursor',
|
||||
'because stale',
|
||||
'terminal worktree',
|
||||
'index snapshot',
|
||||
'reattach cursor',
|
||||
'transaction relay',
|
||||
'snapshot session',
|
||||
'daemon publish',
|
||||
'worktree terminal',
|
||||
'cursor index',
|
||||
'stale relay',
|
||||
'session transaction',
|
||||
'publish snapshot',
|
||||
'reattach daemon'
|
||||
]
|
||||
|
||||
async function indexCorpus(
|
||||
corpus: ToolHeavyCorpus
|
||||
): Promise<{ db: SyncDatabase; release: () => void }> {
|
||||
resetSessionParseCacheForTests()
|
||||
const store = new SessionSearchStore(join(corpus.root, 'index.sqlite'), (error) => {
|
||||
throw error
|
||||
})
|
||||
const unregister = registerSessionSearchIndexConsumer(store)
|
||||
const stats = createSessionParseStats()
|
||||
for (const path of corpus.files) {
|
||||
await parseAgentSessionFileCached(
|
||||
await sessionCandidate('claude', path),
|
||||
process.platform,
|
||||
stats
|
||||
)
|
||||
}
|
||||
return {
|
||||
// The store's own handle, which is what a composed reader gets: every
|
||||
// retrieval is one synchronous statement, so nothing pins a WAL snapshot.
|
||||
db: store.connection,
|
||||
release: () => {
|
||||
unregister()
|
||||
resetTranscriptConsumersForTests()
|
||||
resetSessionParseCacheForTests()
|
||||
store.close()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type Timing = { p50: number; p95: number }
|
||||
|
||||
function timing(samples: readonly number[]): Timing {
|
||||
const sorted = [...samples].sort((left, right) => left - right)
|
||||
const at = (fraction: number): number => {
|
||||
const index = Math.min(sorted.length - 1, Math.floor(sorted.length * fraction))
|
||||
return Math.round((sorted[index] ?? 0) * 100) / 100
|
||||
}
|
||||
return { p50: at(0.5), p95: at(0.95) }
|
||||
}
|
||||
|
||||
/**
|
||||
* The query sets, one per rung of the ladder the engine may take.
|
||||
*
|
||||
* Which rung each one reaches is not forced, it is observed: samples are
|
||||
* bucketed by the route the engine reports, so the table says what was measured
|
||||
* rather than what was intended, and a query that lands on a different rung
|
||||
* than expected shows up as a bucket rather than as a wrong number.
|
||||
*/
|
||||
function queries(): string[] {
|
||||
const run = (index: number, length: number): string =>
|
||||
Array.from({ length }, (_unused, step) => QUERIES[(index + step) % QUERIES.length]).join(' ')
|
||||
return [
|
||||
// Two terms, unquoted: not literal, so straight to OR.
|
||||
...QUERIES,
|
||||
// Two terms, quoted: literal, and on this corpus any two of fourteen words
|
||||
// sit next to each other somewhere, so the phrase rung answers.
|
||||
...QUERIES.map((query) => `"${query}"`),
|
||||
// Eight terms, quoted: an ordered run that long does not occur in 105 MB of
|
||||
// draws from fourteen words, so the phrase rung misses and AND answers.
|
||||
...QUERIES.map((_query, index) => `"${run(index, 4)}"`)
|
||||
]
|
||||
}
|
||||
|
||||
type Bucket = { samples: number[]; hits: number }
|
||||
|
||||
/**
|
||||
* Both scopes over the same queries, interleaved scope by scope: run back to
|
||||
* back, the first one pays for every page the OS cache had not seen and the
|
||||
* ordering moves p95 more than the scope does.
|
||||
*/
|
||||
function scopeReport(db: SyncDatabase): Record<string, unknown> {
|
||||
const engine = new SessionSearchEngine(db)
|
||||
const scopes: SessionSearchScope[] = ['all', 'conversation']
|
||||
const requests: SessionSearchRequest[] = queries().map((query) => ({ query }))
|
||||
const buckets = new Map<string, Bucket>()
|
||||
for (let run = 0; run < WARMUP; run++) {
|
||||
for (const scope of scopes) {
|
||||
for (const request of requests) {
|
||||
engine.search({ ...request, scope })
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const request of requests) {
|
||||
for (const scope of scopes) {
|
||||
const started = performance.now()
|
||||
const result = engine.search({ ...request, scope })
|
||||
const elapsed = performance.now() - started
|
||||
const key = `${result.planner.route}/${scope}`
|
||||
const bucket = buckets.get(key) ?? { samples: [], hits: 0 }
|
||||
bucket.samples.push(elapsed)
|
||||
bucket.hits += result.hits.length
|
||||
buckets.set(key, bucket)
|
||||
}
|
||||
}
|
||||
const report: Record<string, unknown> = {}
|
||||
for (const [key, bucket] of [...buckets].sort(([left], [right]) => left.localeCompare(right))) {
|
||||
report[key] = { ...timing(bucket.samples), samples: bucket.samples.length, hits: bucket.hits }
|
||||
}
|
||||
return report
|
||||
}
|
||||
|
||||
/** Bytes the FTS table occupies, which is the cost the deleted second table saved. */
|
||||
function indexBytes(db: SyncDatabase): Record<string, number> | { unavailable: string } {
|
||||
try {
|
||||
const sum = (where: string, ...values: string[]): number =>
|
||||
Number(
|
||||
(
|
||||
db
|
||||
.prepare(`SELECT COALESCE(SUM(pgsize),0) AS bytes FROM dbstat ${where}`)
|
||||
.get(...values) as { bytes: number }
|
||||
).bytes
|
||||
)
|
||||
return { total: sum(''), messagesFts: sum('WHERE name LIKE ?', 'messages_fts%') }
|
||||
} catch {
|
||||
// dbstat is a compile-time option; the latency numbers stand without it.
|
||||
return { unavailable: 'no dbstat' }
|
||||
}
|
||||
}
|
||||
|
||||
const corpus = await writeToolHeavyCorpus({
|
||||
targetBytes: Number(process.env.CORPUS_MB ?? 100) * 1024 * 1024,
|
||||
toolShare: Number(process.env.TOOL_SHARE ?? 0.9)
|
||||
})
|
||||
let report: string
|
||||
const indexed = await indexCorpus(corpus)
|
||||
try {
|
||||
report = JSON.stringify(
|
||||
{
|
||||
corpus: {
|
||||
sessions: corpus.files.length,
|
||||
transcriptMb: Math.round((corpus.transcriptBytes / 1024 / 1024) * 100) / 100,
|
||||
toolShareOfMessageText:
|
||||
Math.round((corpus.toolBytes / (corpus.toolBytes + corpus.proseBytes)) * 1000) / 1000
|
||||
},
|
||||
indexBytes: indexBytes(indexed.db),
|
||||
route: scopeReport(indexed.db)
|
||||
},
|
||||
null,
|
||||
2
|
||||
)
|
||||
} finally {
|
||||
indexed.release()
|
||||
await rm(corpus.root, { recursive: true, force: true })
|
||||
}
|
||||
|
||||
const out = process.env.BENCH_OUT
|
||||
if (out) {
|
||||
await writeFile(out, `${report}\n`)
|
||||
}
|
||||
console.log(report)
|
||||
@@ -2,9 +2,10 @@ import { mkdtemp, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
|
||||
// The corpus the `conversation_fts` shoot-out runs over. Written here rather
|
||||
// than by `session-search-synthetic-corpus.ts` because the answer turns on one
|
||||
// property that generator fixes: how much of a transcript is tool output.
|
||||
// The corpus the scope benchmark runs over. Written here rather than by
|
||||
// `session-search-synthetic-corpus.ts` because what it costs to answer a
|
||||
// conversation query out of the one FTS table turns on the property that
|
||||
// generator fixes: how much of a transcript is tool output.
|
||||
//
|
||||
// Synthetic, always. This must never be pointed at a real transcript.
|
||||
|
||||
@@ -26,11 +27,10 @@ const PROSE = [
|
||||
]
|
||||
// Tool output is paths, hashes and log lines — and the same words the
|
||||
// conversation uses, because a `rg` over this repository prints them. That
|
||||
// overlap is the whole question: it is what makes a conversation term's posting
|
||||
// list in `messages_fts` far longer than the same term's list in
|
||||
// `conversation_fts`, so the column filter has to read and discard the
|
||||
// difference. A tool vocabulary disjoint from the prose would make the two
|
||||
// tables answer from identical posting lists and prove nothing.
|
||||
// overlap is what the benchmark turns on: it is what makes a conversation
|
||||
// term's posting list carry rows the column filter then has to discard. A tool
|
||||
// vocabulary disjoint from the prose would leave nothing to discard and measure
|
||||
// the wrong thing.
|
||||
const TOOL_ONLY = [
|
||||
'src/main/ai-vault/session-transcript-reader.ts',
|
||||
'node_modules/.pnpm/typescript@5.9.2',
|
||||
@@ -45,10 +45,9 @@ const TOOL_ONLY = [
|
||||
'byteOffset',
|
||||
'MAX_RETRIES'
|
||||
]
|
||||
// Half the tool tokens are conversation words. Deliberately generous to the
|
||||
// table being questioned: the more of a query term lives in `tool_text`, the
|
||||
// more the column filter costs, so a verdict that survives this survives a real
|
||||
// transcript tree.
|
||||
// Half the tool tokens are conversation words. Deliberately pessimistic: the
|
||||
// more of a query term lives in `tool_text`, the more the column filter costs,
|
||||
// so a number measured here holds on a real transcript tree.
|
||||
const TOOL = [...PROSE, ...TOOL_ONLY]
|
||||
|
||||
function mulberry32(seed: number): () => number {
|
||||
|
||||
@@ -44,28 +44,29 @@ several milliseconds run to run if anything else is competing for the disk.
|
||||
|
||||
| Scope | p50 | p95 |
|
||||
| -------------- | ---- | ---- |
|
||||
| `all` | 6.13 | 8.50 |
|
||||
| `conversation` | 3.88 | 5.18 |
|
||||
| `all` | 7.22 | 8.94 |
|
||||
| `conversation` | 5.33 | 7.86 |
|
||||
|
||||
Per query, `all` then `conversation` (p50 / p95):
|
||||
|
||||
| Query | `all` | `conversation` |
|
||||
| --- | --- | --- |
|
||||
| `"terminal reattach"` (phrase) | 3.78 / 4.00 | 2.36 / 2.46 |
|
||||
| `resolveTerminalPath` (identifier) | 6.73 / 6.86 | 4.23 / 4.38 |
|
||||
| `src/main/…/session-transcript-reader.ts` (path) | 8.35 / 9.88 | 4.29 / 4.42 |
|
||||
| `why is the daemon snapshot stale` (prose) | 7.27 / 8.20 | 4.98 / 5.37 |
|
||||
| `reattahc worktre` (typo repair) | 7.42 / 7.82 | 5.07 / 5.18 |
|
||||
| `index` (common term) | 4.70 / 5.53 | 3.28 / 3.66 |
|
||||
| `repo:app-3` (operator only) | 0.11 / 0.13 | 0.10 / 0.14 |
|
||||
| `worktree` scoped to one cwd | 1.34 / 1.40 | 1.02 / 1.06 |
|
||||
| `"terminal reattach"` (phrase) | 5.24 / 8.42 | 2.97 / 3.24 |
|
||||
| `resolveTerminalPath` (identifier) | 7.55 / 8.94 | 6.47 / 6.72 |
|
||||
| `src/main/…/session-transcript-reader.ts` (path) | 8.69 / 10.12 | 7.78 / 8.04 |
|
||||
| `why is the daemon snapshot stale` (prose) | 7.84 / 8.57 | 5.90 / 7.01 |
|
||||
| `reattahc worktre` (typo repair) | 7.30 / 7.39 | 5.53 / 5.89 |
|
||||
| `index` (common term) | 5.45 / 5.66 | 3.81 / 4.02 |
|
||||
| `repo:app-3` (operator only) | 0.12 / 0.16 | 0.10 / 0.10 |
|
||||
| `worktree` scoped to one cwd | 1.47 / 1.63 | 1.25 / 1.49 |
|
||||
|
||||
Reading it:
|
||||
|
||||
- `conversation` is about 1.6x faster at p50 and 1.7x at p95. That gap is the
|
||||
answer to "what is the second table for": it is the corpus a keystroke can
|
||||
afford, and it holds no tool output, so it is also the corpus where a match is
|
||||
something a person wrote.
|
||||
- `conversation` is about 1.4x faster at p50 and 1.1x at p95, and it is a column
|
||||
filter over the same table rather than a table of its own. Narrowing to the
|
||||
two prose columns is what buys the gap: fewer postings to score. It is also
|
||||
the scope where a match is something a person wrote rather than something a
|
||||
tool printed.
|
||||
- A `scopePaths` query is the cheapest real search on the page. It is the one
|
||||
narrowing SQL can express exactly, so it seeks `sessions_cwd_key` and hands
|
||||
ranking a small candidate set.
|
||||
@@ -76,49 +77,57 @@ Reading it:
|
||||
which is one page of that walk; an index where few sessions match the operator
|
||||
will read up to the ceiling in `session-search-retrieval` instead.
|
||||
|
||||
## Two FTS tables, or one with a column filter
|
||||
## What the conversation scope costs at real corpus size
|
||||
|
||||
The stack's open decision 3. `conversation_fts` is a second copy of the two prose
|
||||
columns, and a column filter over `messages_fts` returns **the identical rowid
|
||||
set** — checked here per query, not assumed. So the table exists for latency
|
||||
alone, and this is what that latency is.
|
||||
`conversation` was a second FTS table holding a copy of the two prose columns.
|
||||
It is a column filter now — `{user_text assistant_text}: (…)` with bm25 weights
|
||||
that zero the other two — and PR 2 deleted the table on the strength of the
|
||||
shoot-out this section used to hold: the filter came in at 1.16-1.36x the p95 of
|
||||
the dedicated table, under the 2x bar, while the table cost a tenth of the index
|
||||
to maintain. What follows is what the shipped schema actually does, measured
|
||||
again on the same corpus after the table went and tool rows were capped.
|
||||
|
||||
Corpus: Claude transcripts written by
|
||||
`config/scripts/session-search-conversation-fts-benchmark.ts`, 105 MB, indexed
|
||||
through the real store, at two points in the band a real transcript tree sits in.
|
||||
Half the tokens in tool output are words the conversation also uses, which is
|
||||
deliberately generous to the table under question: the more of a query term lives
|
||||
in `tool_text`, the more the column filter has to read and throw away. Both arms
|
||||
run the engine's own retrieval SQL, differing only in the table, the BM25 weights
|
||||
and the `{user_text assistant_text} :` prefix. Twenty queries per route,
|
||||
interleaved arm by arm, warm cache; two runs.
|
||||
Corpus: Claude transcripts from `config/scripts/session-search-tool-heavy-corpus.ts`,
|
||||
105 MB, indexed through the real store, at two points in the 80-97% band a real
|
||||
transcript tree sits in. Half the tokens in tool output are words the
|
||||
conversation also uses, so a conversation term really does have postings the
|
||||
filter must discard. Twenty queries per rung, both scopes interleaved query by
|
||||
query, warm cache; `config/scripts/session-search-scope-benchmark.ts`, run twice.
|
||||
|
||||
| Tool share of message text | Route | `conversation_fts` p50 / p95 | Column-filtered p50 / p95 | Ratio p95 |
|
||||
| -------------------------- | ------ | ---------------------------- | ------------------------- | --------- |
|
||||
| 86% | phrase | 9.46 / 10.35 | 11.98 / 12.88 | 1.24 |
|
||||
| 86% | and | 16.65 / 17.10 | 19.20 / 19.88 | 1.16 |
|
||||
| 93% | phrase | 4.85 / 5.10 | 6.65 / 6.96 | 1.36 |
|
||||
| 93% | and | 8.39 / 8.86 | 10.23 / 10.67 | 1.20 |
|
||||
| Tool share | Rung | `all` p50 / p95 | `conversation` p50 / p95 |
|
||||
| ---------- | ------ | --------------- | ------------------------ |
|
||||
| 86% | phrase | 16.69 / 17.48 | 13.08 / 13.52 |
|
||||
| 86% | or | 31.91 / 35.74 | 22.25 / 23.87 |
|
||||
| 86% | and | 70.04 / 74.00 | 53.47 / 59.39 |
|
||||
| 93% | phrase | 9.14 / 13.36 | 7.23 / 8.51 |
|
||||
| 93% | or | 16.46 / 18.70 | 12.34 / 14.88 |
|
||||
| 93% | and | 39.65 / 43.44 | 31.05 / 32.92 |
|
||||
|
||||
The second run agreed on every p50 to within 0.2 ms; its one outlier was a 36 ms
|
||||
p95 on the `and` route that hit both arms, which is what 20 samples buys.
|
||||
Three things to read out of it.
|
||||
|
||||
**Verdict: delete it.** The bar was 2x at p95 on the conversation scope, and the
|
||||
column filter comes in at 1.16–1.42x across both shares and both runs, while
|
||||
reading roughly twice the rows to do it (283k unfiltered against 147k filtered on
|
||||
the phrase route at 86%). Deleting it is a schema bump and a rebuild in PR 2, and
|
||||
about ten lines here: `ftsTableFor` returns `messages_fts` for both scopes, the
|
||||
conversation weights become `3.0, 2.0, 0.0, 0.0`, and the expression gains the
|
||||
column prefix. Nothing else depends on the table.
|
||||
**The filter is a win, not a cost.** Every rung is faster narrow than wide, by
|
||||
1.2x to 1.4x at p50. The shoot-out compared the filter against a table built for
|
||||
exactly this query; against the wide table it replaces, it does what the second
|
||||
table did, which is read fewer postings.
|
||||
|
||||
One number argues the other way and is worth stating rather than burying. PR 2
|
||||
priced `conversation_fts` at about a quarter of the index, measured on a corpus
|
||||
whose tool output is 56% of its message text. On a tool-heavy corpus it is **6.7%
|
||||
of the index at 93% tool output and 11% at 86%**, because `messages_fts` grows
|
||||
with the tool text and the second table does not. So the saving is smaller than
|
||||
the decision was framed around, and it is bought with 1.2–1.4x on the latency of
|
||||
the scope a keystroke uses. The threshold says delete; the numbers for keeping it
|
||||
are here so that call can be re-made on sight rather than on memory.
|
||||
**The `and` rung is where the corpus size shows.** Those queries are eight terms,
|
||||
chosen so no ordered run that long occurs and the phrase rung has to miss; a
|
||||
real two-term AND sits nearer the phrase row. It is also the noisiest: the
|
||||
second run's p95 reached 140 ms on one bucket, which is what twenty samples of a
|
||||
70 ms query buys. Read the p50 column.
|
||||
|
||||
**The index is far smaller than the shoot-out's was.** 57 MB at 93% tool output
|
||||
and 103 MB at 86%, against roughly 150 MB for `messages_fts` alone before PR 2
|
||||
capped an indexed tool row at 3,072 characters. Most of a tool-heavy transcript
|
||||
is now not in the index at all, which moves every number above and is the larger
|
||||
effect of the two.
|
||||
|
||||
What is **not** measured here is relevance, and the column filter does carry one
|
||||
ranking difference the deleted table did not. FTS5's bm25 normalises by the
|
||||
whole row's length and has no per-column length, so two rows with identical
|
||||
prose score differently when one also holds tool output. The rowid set is
|
||||
unchanged, which is what the deletion was decided on; the order within it can
|
||||
move. `session-search-engine.test.ts` pins the direction.
|
||||
|
||||
## `sessionCandidateLimit`
|
||||
|
||||
@@ -134,12 +143,12 @@ seen and the ordering alone moves p95 further than the limit does.
|
||||
|
||||
| Limit | p50 | p95 | Pages of 20 a caller can reach |
|
||||
| --- | --- | --- | --- |
|
||||
| 200 | 5.82 | 6.12 | 10 |
|
||||
| 600 | 6.92 | 7.39 | 30 |
|
||||
| 1200 | 8.43 | 9.18 | 60 |
|
||||
| 2400 | 11.30 | 12.18 | 120 |
|
||||
| 200 | 6.85 | 7.21 | 10 |
|
||||
| 600 | 7.93 | 8.36 | 30 |
|
||||
| 1200 | 9.55 | 10.53 | 60 |
|
||||
| 2400 | 12.32 | 13.45 | 120 |
|
||||
|
||||
600 is the default: it costs about 19% over 200 at p50 and buys three times the
|
||||
600 is the default: it costs about 16% over 200 at p50 and buys three times the
|
||||
reachable depth, and the curve only turns steep past 1200. A host with a much
|
||||
larger index can raise it; the result's `truncated.candidates` says when the limit
|
||||
was the thing that cut the answer, so a caller never has to guess.
|
||||
@@ -173,8 +182,13 @@ store keeps answering from the unlinked inode, and this PR is what first makes
|
||||
that reachable, because it is the first thing that reads. What PR 4 does is
|
||||
refuse to make it worse. The engine carries its own schema — the vocabulary, the
|
||||
query log and the generation triggers — and re-creates whatever of it is missing
|
||||
on every search, so a dropped object heals rather than degrading. The one it
|
||||
cannot re-create is the vocabulary's source, because `messages_fts` is the
|
||||
store's; an index in the middle of a rebuild is named as unable to serve typo
|
||||
repair and still answers from the conversation table, rather than throwing on
|
||||
the first query that reaches for one.
|
||||
on every search, so a dropped object heals rather than degrading.
|
||||
|
||||
The one it cannot re-create is the vocabulary's source, because `messages_fts`
|
||||
is the store's. With one FTS table that is also the end of the degrade: there is
|
||||
no second corpus to answer from, so an engine over an index mid-rebuild names
|
||||
typo repair as unavailable and then fails on the table it cannot read, which is
|
||||
the honest outcome — an empty page would read as an answer. `unavailable` can
|
||||
therefore no longer be reported alongside a successful search, and PR 5 should
|
||||
decide whether the field survives into the contract; it becomes reachable again
|
||||
the day something opens the index read-only.
|
||||
|
||||
@@ -50,6 +50,11 @@ export type SyntheticSession = {
|
||||
/** Rows of `text` to write; one session with many rows is one hit. */
|
||||
rows?: number
|
||||
role?: TranscriptMessageRole
|
||||
/**
|
||||
* Written into `tool_text` alongside `text`, which is the one row shape the
|
||||
* conversation scope has to exclude while the `all` scope keeps it.
|
||||
*/
|
||||
toolText?: string
|
||||
agent?: string
|
||||
updatedAt?: string
|
||||
messageCount?: number
|
||||
@@ -67,6 +72,7 @@ export function addSyntheticSession(db: SyncDatabase, session: SyntheticSession)
|
||||
text = 'needle',
|
||||
rows = 1,
|
||||
role = 'user',
|
||||
toolText = '',
|
||||
agent = 'claude',
|
||||
updatedAt = `2026-09-${String((id % 28) + 1).padStart(2, '0')}T00:00:00.000Z`,
|
||||
messageCount = rows,
|
||||
@@ -90,17 +96,10 @@ export function addSyntheticSession(db: SyncDatabase, session: SyntheticSession)
|
||||
)
|
||||
const user = role === 'user' ? text : ''
|
||||
const assistant = role === 'assistant' ? text : ''
|
||||
const tool = role === 'tool' ? text : ''
|
||||
const tool = role === 'tool' ? `${text} ${toolText}`.trim() : toolText
|
||||
db.prepare(
|
||||
'INSERT INTO messages_fts(rowid,user_text,assistant_text,tool_text,identifiers) VALUES (?,?,?,?,?)'
|
||||
).run(messageId, user, assistant, tool, identifierShadowText(text))
|
||||
if (role !== 'tool') {
|
||||
db.prepare('INSERT INTO conversation_fts(rowid,user_text,assistant_text) VALUES (?,?,?)').run(
|
||||
messageId,
|
||||
user,
|
||||
assistant
|
||||
)
|
||||
}
|
||||
).run(messageId, user, assistant, tool, identifierShadowText(`${text} ${toolText}`))
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -20,7 +20,8 @@ export const SESSION_SEARCH_SNIPPET_MARK_CLOSE = ']]'
|
||||
/**
|
||||
* Which corpus answers the query.
|
||||
*
|
||||
* - `conversation`: user and assistant turns only, from `conversation_fts`.
|
||||
* - `conversation`: user and assistant turns only, as a column filter over
|
||||
* `messages_fts` (see `scopedExpression`).
|
||||
* - `all`: those turns plus tool calls and tool output, and the identifier
|
||||
* shadow column, from `messages_fts`.
|
||||
*
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { SESSION_SEARCH_QUERY_MAX_LENGTH } from './session-search-engine-types'
|
||||
import type { SessionSearchRequest, SessionSearchResponse } from './session-search-engine-types'
|
||||
import { planSessionSearchQuery } from './session-search-query-planner'
|
||||
import { ensureSessionSearchQuerySchema } from './session-search-query-schema'
|
||||
import { EMPTY_SNIPPET, sessionSearchSnippet } from './session-search-snippet'
|
||||
import {
|
||||
addSyntheticSession,
|
||||
markFork,
|
||||
@@ -132,6 +135,60 @@ describe('scope picks the corpus and never switches it', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('the conversation scope is a column filter, and it binds the whole query', () => {
|
||||
it('refuses an AND whose second term lives only in tool output', async () => {
|
||||
// The filter binds to the expression it prefixes. `{cols}: (a AND b)`
|
||||
// filters both terms; `{cols}: a AND b` filters only `a` and searches tool
|
||||
// output for the rest, which is a conversation search answering from a
|
||||
// column it promised not to read.
|
||||
const { db, engine } = await open('ss-engine-scope-binding')
|
||||
addSyntheticSession(db, { id: 1, text: 'alpha gamma beta' })
|
||||
addSyntheticSession(db, { id: 2, text: 'alpha gamma', toolText: 'beta' })
|
||||
// Quoted, so the query is literal; not adjacent, so the phrase rung misses
|
||||
// and the AND rung is the one that answers.
|
||||
const query = '"alpha" beta'
|
||||
|
||||
const wide = engine.search({ query, scope: 'all' })
|
||||
expect(wide.planner.route).toBe('and')
|
||||
expect(ids(wide).sort()).toEqual(['1', '2'])
|
||||
|
||||
const narrowed = engine.search({ query, scope: 'conversation' })
|
||||
expect(narrowed.planner.route).toBe('and')
|
||||
expect(ids(narrowed)).toEqual(['1'])
|
||||
})
|
||||
|
||||
it('ranks a conversation hit down for tool output it will not show', async () => {
|
||||
// The one behavioural difference the column filter carries, pinned rather
|
||||
// than wished away. FTS5's bm25 normalises by the whole row's length and
|
||||
// has no per-column length, so two rows with identical prose do not score
|
||||
// identically when one of them also holds tool output. A dedicated
|
||||
// two-column table scored them the same. The rowid set is unchanged, which
|
||||
// is what the decision was measured on; the order within it can move.
|
||||
const { db, engine } = await open('ss-engine-scope-weights')
|
||||
addSyntheticSession(db, { id: 1, text: 'harbor pilot' })
|
||||
addSyntheticSession(db, { id: 2, text: 'harbor pilot', toolText: 'unrelated '.repeat(40) })
|
||||
const narrowed = engine.search({ query: 'harbor', scope: 'conversation' })
|
||||
expect(ids(narrowed)).toEqual(['1', '2'])
|
||||
expect(narrowed.hits[0]!.score).toBeGreaterThan(narrowed.hits[1]!.score)
|
||||
})
|
||||
|
||||
it('never snippets a conversation hit out of tool output', async () => {
|
||||
const { db, engine } = await open('ss-engine-scope-snippet')
|
||||
addSyntheticSession(db, { id: 1, text: 'harbor pilot', toolText: 'harbor tool output line' })
|
||||
const [hit] = engine.search({ query: 'harbor', scope: 'conversation' }).hits
|
||||
expect(hit?.evidence?.snippet).toContain('pilot')
|
||||
expect(hit?.evidence?.snippet).not.toContain('output')
|
||||
// And asked for a tool-only row directly, it has nothing to show.
|
||||
addSyntheticSession(db, { id: 2, text: 'harbor tool output line', role: 'tool' })
|
||||
const rowid = Number(
|
||||
(db.prepare('SELECT max(id) AS id FROM messages').get() as { id: number }).id
|
||||
)
|
||||
const plan = planSessionSearchQuery('harbor')
|
||||
expect(sessionSearchSnippet(db, 'conversation', rowid, plan)).toEqual(EMPTY_SNIPPET)
|
||||
expect(sessionSearchSnippet(db, 'all', rowid, plan).text).toContain('output')
|
||||
})
|
||||
})
|
||||
|
||||
describe('a session is one hit, however many of its rows matched', () => {
|
||||
it.each(['relevance', 'newest'] as const)(
|
||||
'keeps a short session on the %s page beside a 650-row session',
|
||||
@@ -295,23 +352,22 @@ describe('the engine carries its own schema and puts it back', () => {
|
||||
})
|
||||
|
||||
it('names the feature it cannot serve when the vocabulary has no source left', async () => {
|
||||
// What an index being rebuilt by another handle looks like from here: the
|
||||
// What an index being rebuilt by another handle looks like from here. The
|
||||
// vocabulary can be created over a missing `messages_fts` and every query
|
||||
// against it then fails, so the engine reads the source, not the view.
|
||||
// The conversation scope has its own table and keeps answering.
|
||||
// against it then fails, so the probe reads the source, not the view.
|
||||
//
|
||||
// With one FTS table there is no scope left to answer from, so this is now
|
||||
// the boundary of the degrade: the engine names the feature and the search
|
||||
// fails loudly on the table it cannot read, rather than returning an empty
|
||||
// page that looks like an answer.
|
||||
const { db, engine } = await open('ss-engine-vocab-source-gone')
|
||||
addSyntheticSession(db, { id: 1, text: 'coalesces here now', role: 'user' })
|
||||
addSyntheticSession(db, { id: 2, text: 'coalesces again here', role: 'user' })
|
||||
db.exec('DROP TABLE messages_vocab; DROP TABLE messages_fts')
|
||||
|
||||
const result = engine.search({ query: 'coalescs', scope: 'conversation' })
|
||||
expect(result.unavailable).toEqual(['typo-repair'])
|
||||
expect(result.planner.route).toBe('or')
|
||||
expect(result.hits).toEqual([])
|
||||
expect(ids(engine.search({ query: 'coalesces', scope: 'conversation' })).sort()).toEqual([
|
||||
'1',
|
||||
'2'
|
||||
])
|
||||
expect(ensureSessionSearchQuerySchema(db)).toEqual(['typo-repair'])
|
||||
for (const scope of ['all', 'conversation'] as const) {
|
||||
expect(() => engine.search({ query: 'coalesces', scope })).toThrow(/no such (fts5 )?table/i)
|
||||
}
|
||||
})
|
||||
|
||||
it('picks the feature back up when the source comes back', async () => {
|
||||
@@ -324,9 +380,7 @@ describe('the engine carries its own schema and puts it back', () => {
|
||||
}
|
||||
).sql
|
||||
db.exec('DROP TABLE messages_vocab; DROP TABLE messages_fts')
|
||||
expect(engine.search({ query: 'coalescs', scope: 'conversation' }).unavailable).toEqual([
|
||||
'typo-repair'
|
||||
])
|
||||
expect(ensureSessionSearchQuerySchema(db)).toEqual(['typo-repair'])
|
||||
|
||||
db.exec(fts)
|
||||
// Two, because the vocabulary only offers a term at least two rows carry.
|
||||
|
||||
@@ -12,6 +12,7 @@ import {
|
||||
type SessionSearchHit,
|
||||
type SessionSearchRequest,
|
||||
type SessionSearchResponse,
|
||||
type SessionSearchScope,
|
||||
type SessionSearchSourcePresence
|
||||
} from './session-search-engine-types'
|
||||
import { readIndexGeneration } from './session-search-index-generation'
|
||||
@@ -29,7 +30,6 @@ import {
|
||||
import { planSessionSearchQuery } from './session-search-query-planner'
|
||||
import { logSessionSearchQuery } from './session-search-query-log'
|
||||
import {
|
||||
ftsTableFor,
|
||||
SessionSearchRetrieval,
|
||||
type RetrievalScope,
|
||||
type Retrieved
|
||||
@@ -149,7 +149,7 @@ export class SessionSearchEngine {
|
||||
|
||||
const limit = resolveSessionSearchLimit(request.limit)
|
||||
const page = ranked.slice(offset, offset + limit)
|
||||
const hits = this.hits(page, ftsTableFor(scope), retrieved)
|
||||
const hits = this.hits(page, scope, retrieved)
|
||||
const hasMore = ranked.length > offset + limit
|
||||
const response: SessionSearchResponse = {
|
||||
hits,
|
||||
@@ -264,26 +264,26 @@ export class SessionSearchEngine {
|
||||
/** Snippets and source presence are paid for by the page, never by the list. */
|
||||
private hits(
|
||||
page: readonly RankedSession[],
|
||||
table: 'messages_fts' | 'conversation_fts',
|
||||
scope: SessionSearchScope,
|
||||
retrieved: Retrieved | null
|
||||
): SessionSearchHit[] {
|
||||
const presence = sessionSourcePresence(
|
||||
this.db,
|
||||
page.map((entry) => entry.session.id)
|
||||
)
|
||||
return page.map((entry) => this.hit(entry, table, retrieved, presence))
|
||||
return page.map((entry) => this.hit(entry, scope, retrieved, presence))
|
||||
}
|
||||
|
||||
private hit(
|
||||
entry: RankedSession,
|
||||
table: 'messages_fts' | 'conversation_fts',
|
||||
scope: SessionSearchScope,
|
||||
retrieved: Retrieved | null,
|
||||
presence: ReadonlyMap<number, SessionSearchSourcePresence>
|
||||
): SessionSearchHit {
|
||||
const { session, message } = entry
|
||||
const snippet =
|
||||
message && retrieved
|
||||
? sessionSearchSnippet(this.db, table, message.rowid, retrieved.plan)
|
||||
? sessionSearchSnippet(this.db, scope, message.rowid, retrieved.plan)
|
||||
: EMPTY_SNIPPET
|
||||
return {
|
||||
...sessionFields(session),
|
||||
|
||||
@@ -45,11 +45,6 @@ function plantOrphans(db: SyncDatabase): number[] {
|
||||
db.prepare(
|
||||
'INSERT INTO messages_fts(rowid,user_text,assistant_text,tool_text,identifiers) VALUES (?,?,?,?,?)'
|
||||
).run(rowid, ORPHAN_TEXT, '', '', identifierShadowText(ORPHAN_TEXT))
|
||||
db.prepare('INSERT INTO conversation_fts(rowid,user_text,assistant_text) VALUES (?,?,?)').run(
|
||||
rowid,
|
||||
ORPHAN_TEXT,
|
||||
''
|
||||
)
|
||||
rowids.push(rowid)
|
||||
}
|
||||
return rowids
|
||||
@@ -98,8 +93,8 @@ it('never repairs a term onto a spelling only orphaned rows carry', async () =>
|
||||
it('snippets nothing for an orphaned row, even asked for it by rowid', async () => {
|
||||
const { harness: open, rowids } = await withOrphans()
|
||||
const plan = planSessionSearchQuery('marmoset')
|
||||
for (const table of ['messages_fts', 'conversation_fts'] as const) {
|
||||
expect(sessionSearchSnippet(open.db, table, rowids[0]!, plan)).toEqual({
|
||||
for (const scope of ['all', 'conversation'] as const) {
|
||||
expect(sessionSearchSnippet(open.db, scope, rowids[0]!, plan)).toEqual({
|
||||
text: '',
|
||||
truncated: false
|
||||
})
|
||||
|
||||
@@ -18,7 +18,13 @@ const RECENT_SCAN_FACTOR = 20
|
||||
|
||||
// Measured: user 3 / assistant 2 / tool 1 / identifiers 1 (MRR 0.503 vs 0.475 flat).
|
||||
const FULL_WEIGHTS = '3.0, 2.0, 1.0, 1.0'
|
||||
const CONVERSATION_WEIGHTS = '3.0, 2.0'
|
||||
// The conversation scope zeroes the two columns its filter already excludes.
|
||||
// Measured, and stated because it is easy to over-read: these zeros change no
|
||||
// score. FTS5's bm25 sums over the columns the query matched, and the filter
|
||||
// has already kept the match out of those two, so the same rows come back with
|
||||
// `1.0, 1.0` here. They are a statement of what the scope means, not the fence
|
||||
// that enforces it — `scopedExpression` is the fence.
|
||||
const CONVERSATION_WEIGHTS = '3.0, 2.0, 0.0, 0.0'
|
||||
|
||||
export type RetrievalScope = {
|
||||
scope: SessionSearchScope
|
||||
@@ -46,8 +52,27 @@ export type Retrieved = {
|
||||
repairedTerms?: string[]
|
||||
}
|
||||
|
||||
export function ftsTableFor(scope: SessionSearchScope): 'messages_fts' | 'conversation_fts' {
|
||||
return scope === 'all' ? 'messages_fts' : 'conversation_fts'
|
||||
/**
|
||||
* What a scope is, now that there is one FTS table.
|
||||
*
|
||||
* `conversation` used to be a second table holding a copy of the two prose
|
||||
* columns. It is a column filter instead: PR 2 measured the filter at
|
||||
* 1.16-1.36x the p95 of the dedicated table on a 105 MB corpus, against a 2x
|
||||
* bar, and the table cost a tenth of the index to maintain.
|
||||
*
|
||||
* The filter binds to the whole expression, so it is applied here and nowhere
|
||||
* else — `{cols}: (a AND b)` filters both terms, while a prefix pasted in front
|
||||
* of a bare `a AND b` would filter only `a` and quietly search tool output for
|
||||
* the rest.
|
||||
*/
|
||||
const CONVERSATION_COLUMNS = '{user_text assistant_text}'
|
||||
|
||||
export function scopedWeights(scope: SessionSearchScope): string {
|
||||
return scope === 'all' ? FULL_WEIGHTS : CONVERSATION_WEIGHTS
|
||||
}
|
||||
|
||||
export function scopedExpression(scope: SessionSearchScope, expression: string): string {
|
||||
return scope === 'all' ? expression : `${CONVERSATION_COLUMNS}: (${expression})`
|
||||
}
|
||||
|
||||
/** The FTS half of a search: the route ladder and the SQL each rung runs. */
|
||||
@@ -193,12 +218,11 @@ export class SessionSearchRetrieval {
|
||||
const eligible = filter.conditions.length
|
||||
? ` AND m.session_row_id IN (SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')})`
|
||||
: ''
|
||||
const table = ftsTableFor(scope.scope)
|
||||
const weights = scope.scope === 'all' ? FULL_WEIGHTS : CONVERSATION_WEIGHTS
|
||||
const matched = `SELECT ${table}.rowid AS rowid, -bm25(${table}, ${weights}) AS score,
|
||||
const matched = `SELECT messages_fts.rowid AS rowid,
|
||||
-bm25(messages_fts, ${scopedWeights(scope.scope)}) AS score,
|
||||
m.session_row_id, m.role, m.ts, s.updated_at
|
||||
FROM ${table} JOIN messages m ON m.id = ${table}.rowid
|
||||
JOIN sessions s ON s.id = m.session_row_id WHERE ${table} MATCH ?${eligible}`
|
||||
FROM messages_fts JOIN messages m ON m.id = messages_fts.rowid
|
||||
JOIN sessions s ON s.id = m.session_row_id WHERE messages_fts MATCH ?${eligible}`
|
||||
// Why: collapse to one row per session BEFORE the candidate limit, on both
|
||||
// sort orders, so a single long session cannot occupy the whole page.
|
||||
// `max(score)` makes SQLite pick that session's best row for the bare columns.
|
||||
@@ -210,6 +234,8 @@ export class SessionSearchRetrieval {
|
||||
const sql = `WITH matched AS MATERIALIZED (${matched})
|
||||
SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched
|
||||
GROUP BY session_row_id ORDER BY ${order} LIMIT ${candidateLimit}`
|
||||
return this.db.prepare(sql).all(expression, ...filter.values) as MessageRow[]
|
||||
return this.db
|
||||
.prepare(sql)
|
||||
.all(scopedExpression(scope.scope, expression), ...filter.values) as MessageRow[]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,8 @@ import {
|
||||
SESSION_SEARCH_SNIPPET_MARK_OPEN
|
||||
} from './session-search-engine-types'
|
||||
import { orExpression, type SessionSearchQueryPlan } from './session-search-query-planner'
|
||||
import { scopedExpression } from './session-search-retrieval'
|
||||
import type { SessionSearchScope } from './session-search-engine-types'
|
||||
|
||||
const SNIPPET_TOKENS = 12
|
||||
// Why a ceiling on top of the token count: a transcript chunk can be 8000
|
||||
@@ -27,18 +29,26 @@ export const EMPTY_SNIPPET: SessionSearchSnippet = { text: '', truncated: false
|
||||
*/
|
||||
export function sessionSearchSnippet(
|
||||
db: SyncDatabase,
|
||||
table: 'messages_fts' | 'conversation_fts',
|
||||
scope: SessionSearchScope,
|
||||
rowid: number,
|
||||
plan: SessionSearchQueryPlan
|
||||
): SessionSearchSnippet {
|
||||
// Why: the identifier shadow column is word soup; a hit that also matches in a
|
||||
// prose column should be shown from there. Column -1 (any column) is the
|
||||
// fallback for rows that only matched through the shadow column.
|
||||
const columns = table === 'messages_fts' ? [0, 1, 2, -1] : [0, 1, -1]
|
||||
//
|
||||
// The same four for every scope, because the scope is already in the
|
||||
// expression below. A conversation snippet cannot come out of `tool_text` for
|
||||
// the reason the search could not: the row has to match
|
||||
// `{user_text assistant_text}: …` before any of these columns is read, and a
|
||||
// row that matches under that filter carries its mark in column 0 or 1. A
|
||||
// second list here would be a guard with nothing left to guard, and the two
|
||||
// would mask each other's mistakes.
|
||||
const columns = [0, 1, 2, -1]
|
||||
const select = columns
|
||||
.map(
|
||||
(column, index) =>
|
||||
`snippet(${table}, ${column}, '${SESSION_SEARCH_SNIPPET_MARK_OPEN}', '${SESSION_SEARCH_SNIPPET_MARK_CLOSE}', '…', ${SNIPPET_TOKENS}) AS c${index}`
|
||||
`snippet(messages_fts, ${column}, '${SESSION_SEARCH_SNIPPET_MARK_OPEN}', '${SESSION_SEARCH_SNIPPET_MARK_CLOSE}', '…', ${SNIPPET_TOKENS}) AS c${index}`
|
||||
)
|
||||
.join(', ')
|
||||
try {
|
||||
@@ -51,12 +61,14 @@ export function sessionSearchSnippet(
|
||||
// returned to a caller.
|
||||
const row = db
|
||||
.prepare(
|
||||
`SELECT ${select} FROM ${table}
|
||||
JOIN messages m ON m.id = ${table}.rowid
|
||||
`SELECT ${select} FROM messages_fts
|
||||
JOIN messages m ON m.id = messages_fts.rowid
|
||||
JOIN sessions s ON s.id = m.session_row_id
|
||||
WHERE ${table} MATCH ? AND ${table}.rowid IN (SELECT ?)`
|
||||
WHERE messages_fts MATCH ? AND messages_fts.rowid IN (SELECT ?)`
|
||||
)
|
||||
.get(orExpression(plan.terms), rowid) as Record<string, string> | undefined
|
||||
.get(scopedExpression(scope, orExpression(plan.terms)), rowid) as
|
||||
| Record<string, string>
|
||||
| undefined
|
||||
if (!row) {
|
||||
return EMPTY_SNIPPET
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user