diff --git a/.gitignore b/.gitignore index 913dfc4a045..5bb1fedcaee 100644 --- a/.gitignore +++ b/.gitignore @@ -103,6 +103,7 @@ docs/** !docs/agent-skill-sharing-implementation-checklist.md !docs/mobile-terminal-shortcut-bar.md !docs/reference/ +!docs/reference/agent-session-search-query-tuning.md !docs/reference/agent-status-store.md !docs/reference/git-compatibility.md !docs/reference/headless-linux-server.md diff --git a/config/scripts/session-search-query-benchmark.ts b/config/scripts/session-search-query-benchmark.ts new file mode 100644 index 00000000000..bacfe953f02 --- /dev/null +++ b/config/scripts/session-search-query-benchmark.ts @@ -0,0 +1,197 @@ +import { rm, writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { + createSessionParseStats, + parseAgentSessionFileCached, + resetSessionParseCacheForTests +} from '../../src/main/ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers' +import { SessionSearchEngine } from '../../src/main/ai-vault-search/session-search-engine' +import type { + SessionSearchRequest, + SessionSearchScope +} from '../../src/main/ai-vault-search/session-search-engine-types' +import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer' +import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' +import { + writeSyntheticTranscriptCorpus, + type SyntheticCorpus, + type SyntheticCorpusOptions +} from '../../src/main/ai-vault-search/session-search-synthetic-corpus' +import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures' + +// What a query costs, and what the session candidate limit buys. Everything +// runs through the real store and the real engine over a synthetic corpus; +// never point this at a real transcript tree. + +const WARMUP = 5 +const SAMPLES = 25 + +// One query per rung the ladder can take, plus the two shapes that skip it. +const QUERIES: { name: string; request: SessionSearchRequest }[] = [ + { name: 'phrase', request: { query: '"terminal reattach"' } }, + { name: 'identifier', request: { query: 'resolveTerminalPath' } }, + { name: 'path', request: { query: 'src/main/ai-vault/session-transcript-reader.ts' } }, + { name: 'prose', request: { query: 'why is the daemon snapshot stale' } }, + { name: 'typo', request: { query: 'reattahc worktre' } }, + { name: 'common-term', request: { query: 'index' } }, + { name: 'operator-only', request: { query: 'repo:app-3' } }, + { name: 'scoped', request: { query: 'worktree', filters: { scopePaths: ['/repo/app-3'] } } } +] + +type Timing = { p50: number; p95: number } + +function percentile(sorted: readonly number[], fraction: number): number { + const at = Math.min(sorted.length - 1, Math.floor(sorted.length * fraction)) + return Math.round((sorted[at] ?? 0) * 100) / 100 +} + +function timing(samples: number[]): Timing { + const sorted = [...samples].sort((left, right) => left - right) + return { p50: percentile(sorted, 0.5), p95: percentile(sorted, 0.95) } +} + +function time(engine: SessionSearchEngine, request: SessionSearchRequest): number { + const started = performance.now() + engine.search(request) + return performance.now() - started +} + +async function indexCorpus( + options: SyntheticCorpusOptions +): Promise<{ corpus: SyntheticCorpus; store: SessionSearchStore; release: () => void }> { + resetSessionParseCacheForTests() + const corpus = await writeSyntheticTranscriptCorpus(options) + const store = new SessionSearchStore(join(corpus.root, 'index.sqlite'), (error) => { + throw error + }) + const unregister = registerSessionSearchIndexConsumer(store) + const stats = createSessionParseStats() + for (const path of corpus.files) { + await parseAgentSessionFileCached( + await sessionCandidate('claude', path), + process.platform, + stats + ) + } + await store.settled() + await store.warm() + return { + corpus, + store, + release: () => { + unregister() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store.close() + } + } +} + +/** Per-query and overall latency for one scope. */ +function scopeReport( + store: SessionSearchStore, + scope: SessionSearchScope +): Record { + const engine = new SessionSearchEngine(store) + const everything: number[] = [] + const perQuery: Record = {} + for (const { name, request } of QUERIES) { + const scoped = { ...request, scope } + for (let run = 0; run < WARMUP; run++) { + engine.search(scoped) + } + const samples = Array.from({ length: SAMPLES }, () => time(engine, scoped)) + everything.push(...samples) + const result = engine.search(scoped) + perQuery[name] = { ...timing(samples), hits: result.hits.length, route: result.planner.route } + } + return { ...timing(everything), perQuery } +} + +/** + * The candidate limit only costs anything once there are more matching sessions + * than the limit, so this runs over many short sessions rather than the wide + * corpus above. Limits are interleaved sample by sample: run back to back, the + * first configuration pays for every page the OS cache had not seen yet and the + * ordering alone moves p95 by more than the limit does. + */ +function candidateSweep( + store: SessionSearchStore, + limits: readonly number[] +): Record { + const request: SessionSearchRequest = { query: 'index', limit: 20 } + const engines = new Map( + limits.map((limit) => [limit, new SessionSearchEngine(store, { sessionCandidateLimit: limit })]) + ) + const samples = new Map(limits.map((limit) => [limit, [] as number[]])) + for (let run = 0; run < WARMUP; run++) { + for (const engine of engines.values()) { + engine.search(request) + } + } + for (let run = 0; run < SAMPLES; run++) { + for (const limit of limits) { + samples.get(limit)!.push(time(engines.get(limit)!, request)) + } + } + const report: Record = {} + for (const limit of limits) { + const result = engines.get(limit)!.search(request) + report[String(limit)] = { + ...timing(samples.get(limit)!), + truncated: result.truncated.candidates, + // Pages a caller could walk before the limit stops handing out sessions. + reachablePages: Math.ceil(Math.min(limit, 20 * 1000) / 20) + } + } + return report +} + +const wide = await indexCorpus({ sessions: Number(process.env.SESSIONS ?? 40) }) +let report: string +try { + const scope = { + all: scopeReport(wide.store, 'all'), + conversation: scopeReport(wide.store, 'conversation') + } + wide.release() + await rm(wide.corpus.root, { recursive: true, force: true }) + + // Many short sessions: what makes the candidate limit binding is the session + // count, not the byte count. + const many = await indexCorpus({ sessions: 2500, turnsPerSession: 1, seed: 7 }) + try { + report = JSON.stringify( + { + scopeCorpus: { + sessions: wide.corpus.files.length, + transcriptMb: Math.round((wide.corpus.transcriptBytes / 1024 / 1024) * 100) / 100, + messages: wide.corpus.messageCount + }, + scope, + candidateCorpus: { + sessions: many.corpus.files.length, + transcriptMb: Math.round((many.corpus.transcriptBytes / 1024 / 1024) * 100) / 100 + }, + candidateSweep: candidateSweep(many.store, [200, 600, 1200, 2400]) + }, + null, + 2 + ) + } finally { + many.release() + await rm(many.corpus.root, { recursive: true, force: true }) + } +} catch (error) { + await rm(wide.corpus.root, { recursive: true, force: true }) + throw error +} + +// Why a file as well as stdout: a runner that intercepts console output +// (vitest does) would otherwise swallow the whole report. +const out = process.env.BENCH_OUT +if (out) { + await writeFile(out, `${report}\n`) +} +console.log(report) diff --git a/docs/reference/agent-session-search-query-tuning.md b/docs/reference/agent-session-search-query-tuning.md new file mode 100644 index 00000000000..97bffe96619 --- /dev/null +++ b/docs/reference/agent-session-search-query-tuning.md @@ -0,0 +1,94 @@ +# Agent session search: query tuning + +What a search costs, and what the knobs in `src/main/ai-vault-search/session-search-engine.ts` +buy. Every number here comes from `config/scripts/session-search-query-benchmark.ts` +over the synthetic corpus in `session-search-synthetic-corpus.ts`. Nothing in this +file was measured against a real transcript, and the benchmark must never be +pointed at one. + +## Running it + +The benchmark is a top-level-await module that imports the main-process tree by +extensionless path, so it needs a bundler-backed runner rather than bare `node`: + +```sh +cat > src/main/ai-vault-search/zz-bench.test.ts <<'EOF' +import { it } from 'vitest' +it('runs', { timeout: 600_000 }, async () => { + await import('../../../config/scripts/session-search-query-benchmark') +}) +EOF +BENCH_OUT=/tmp/ss-query-bench.json pnpm test src/main/ai-vault-search/zz-bench.test.ts +rm src/main/ai-vault-search/zz-bench.test.ts +``` + +`BENCH_OUT` exists because vitest intercepts `console.log`; the report is written +to that path as well as printed. + +## Scope: what the second FTS table buys a reader + +Corpus: 40 synthetic Claude transcripts, 10.5 MB, 9,600 messages, indexed through +the real store. Eight queries, one per rung of the route ladder plus the two +shapes that skip it; 5 warm-up runs and 25 samples each. Apple silicon, warm page +cache. Milliseconds. + +| Scope | p50 | p95 | +| -------------- | ---- | ----- | +| `all` | 8.46 | 41.74 | +| `conversation` | 5.32 | 12.20 | + +Per query, `all` then `conversation` (p50 / p95): + +| Query | `all` | `conversation` | +| ------------------------------------------------ | ------------- | -------------- | +| `"terminal reattach"` (phrase) | 5.03 / 5.86 | 2.77 / 3.78 | +| `resolveTerminalPath` (identifier) | 8.47 / 9.54 | 5.39 / 17.89 | +| `src/main/…/session-transcript-reader.ts` (path) | 11.96 / 39.31 | 5.55 / 17.27 | +| `why is the daemon snapshot stale` (prose) | 10.78 / 25.23 | 7.11 / 12.34 | +| `reattahc worktre` (typo repair) | 10.41 / 115.05| 6.73 / 9.41 | +| `index` (common term) | 7.25 / 29.99 | 4.98 / 11.51 | +| `repo:app-3` (operator only) | 0.10 / 0.60 | 0.10 / 0.20 | +| `worktree` scoped to one cwd | 2.10 / 2.60 | 1.48 / 2.06 | + +Reading it: + +- `conversation` is about 1.6x faster at p50 and 3.4x at p95. That gap is the + answer to "what is the second table for": it is the corpus a keystroke can + afford, and it holds no tool output, so it is also the corpus where a match is + something a person wrote. +- An operator-only query never touches FTS at all. It is a range seek on + `sessions_cwd_key`, and it costs a tenth of a millisecond. +- Typo repair's p95 in `all` is the worst number on the page. The repair walks + `messages_vocab` per prefix, and the first walk after a cold statement cache + pays for the b-tree pages. It is a first-query cost, not a per-query one. + +## `sessionCandidateLimit` + +The reviewer's F13: this is a tunable default, not a constant. It bounds how many +sessions the SQL hands ranking, so it bounds both retrieval cost and how deep a +caller can page before the answer simply stops. + +The limit only costs anything once more sessions match than the limit allows, so +this is measured over a second corpus: 2,500 one-turn transcripts, 10.9 MB, every +one of them matching the query. Limits are interleaved sample by sample, because +run back to back the first configuration pays for every page the OS cache had not +seen and the ordering alone moves p95 further than the limit does. + +| Limit | p50 | p95 | Pages of 20 a caller can reach | +| ----- | ----- | ----- | ------------------------------ | +| 200 | 9.50 | 13.18 | 10 | +| 600 | 10.86 | 20.30 | 30 | +| 1200 | 12.65 | 15.20 | 60 | +| 2400 | 17.68 | 27.08 | 120 | + +600 is the default: it costs about 14% over 200 at p50 and buys three times the +reachable depth, and the curve only turns steep past 1200. A host with a much +larger index can raise it; the result's `truncated.candidates` says when the limit +was the thing that cut the answer, so a caller never has to guess. + +What is **not** measured here is relevance. These numbers say what a limit costs, +not what it retrieves. The MRR figures quoted in the BM25 weights +(`session-search-retrieval.ts`) and in the identifier shadow column +(`session-search-identifier-split.ts`) come from the original retrieval shoot-out +on real transcripts and are not reproducible from this repository. Any change to +the limit justified on relevance grounds needs an eval set, not this benchmark.