mirror of
https://github.com/stablyai/orca.git
synced 2026-10-03 00:02:19 +00:00
perf(ai-vault-search): measure what a query costs and what the candidate limit buys
Two corpora, because they answer different questions. The 10.5 MB / 40-session corpus is what the scope split costs a reader: conversation is about 1.6x faster at p50 and 3.4x at p95 than the full corpus, which is the argument for the second FTS table being the one a keystroke can afford. The candidate limit needs more sessions than the limit before it costs anything, so it is swept over 2,500 one-turn transcripts with every session matching. Limits are interleaved sample by sample: run back to back, the first configuration pays for every page the OS cache had not seen and the ordering alone moved p95 further than the limit did. The doc says plainly what these numbers do not cover. They are cost, not relevance; the MRR figures quoted beside the BM25 weights and the identifier shadow column come from a shoot-out over real transcripts and cannot be reproduced from this repository.
This commit is contained in:
@@ -103,6 +103,7 @@ docs/**
|
||||
!docs/agent-skill-sharing-implementation-checklist.md
|
||||
!docs/mobile-terminal-shortcut-bar.md
|
||||
!docs/reference/
|
||||
!docs/reference/agent-session-search-query-tuning.md
|
||||
!docs/reference/agent-status-store.md
|
||||
!docs/reference/git-compatibility.md
|
||||
!docs/reference/headless-linux-server.md
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
import { rm, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import {
|
||||
createSessionParseStats,
|
||||
parseAgentSessionFileCached,
|
||||
resetSessionParseCacheForTests
|
||||
} from '../../src/main/ai-vault/session-scanner-parse-cache'
|
||||
import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers'
|
||||
import { SessionSearchEngine } from '../../src/main/ai-vault-search/session-search-engine'
|
||||
import type {
|
||||
SessionSearchRequest,
|
||||
SessionSearchScope
|
||||
} from '../../src/main/ai-vault-search/session-search-engine-types'
|
||||
import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import {
|
||||
writeSyntheticTranscriptCorpus,
|
||||
type SyntheticCorpus,
|
||||
type SyntheticCorpusOptions
|
||||
} from '../../src/main/ai-vault-search/session-search-synthetic-corpus'
|
||||
import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures'
|
||||
|
||||
// What a query costs, and what the session candidate limit buys. Everything
|
||||
// runs through the real store and the real engine over a synthetic corpus;
|
||||
// never point this at a real transcript tree.
|
||||
|
||||
const WARMUP = 5
|
||||
const SAMPLES = 25
|
||||
|
||||
// One query per rung the ladder can take, plus the two shapes that skip it.
|
||||
const QUERIES: { name: string; request: SessionSearchRequest }[] = [
|
||||
{ name: 'phrase', request: { query: '"terminal reattach"' } },
|
||||
{ name: 'identifier', request: { query: 'resolveTerminalPath' } },
|
||||
{ name: 'path', request: { query: 'src/main/ai-vault/session-transcript-reader.ts' } },
|
||||
{ name: 'prose', request: { query: 'why is the daemon snapshot stale' } },
|
||||
{ name: 'typo', request: { query: 'reattahc worktre' } },
|
||||
{ name: 'common-term', request: { query: 'index' } },
|
||||
{ name: 'operator-only', request: { query: 'repo:app-3' } },
|
||||
{ name: 'scoped', request: { query: 'worktree', filters: { scopePaths: ['/repo/app-3'] } } }
|
||||
]
|
||||
|
||||
type Timing = { p50: number; p95: number }
|
||||
|
||||
function percentile(sorted: readonly number[], fraction: number): number {
|
||||
const at = Math.min(sorted.length - 1, Math.floor(sorted.length * fraction))
|
||||
return Math.round((sorted[at] ?? 0) * 100) / 100
|
||||
}
|
||||
|
||||
function timing(samples: number[]): Timing {
|
||||
const sorted = [...samples].sort((left, right) => left - right)
|
||||
return { p50: percentile(sorted, 0.5), p95: percentile(sorted, 0.95) }
|
||||
}
|
||||
|
||||
function time(engine: SessionSearchEngine, request: SessionSearchRequest): number {
|
||||
const started = performance.now()
|
||||
engine.search(request)
|
||||
return performance.now() - started
|
||||
}
|
||||
|
||||
async function indexCorpus(
|
||||
options: SyntheticCorpusOptions
|
||||
): Promise<{ corpus: SyntheticCorpus; store: SessionSearchStore; release: () => void }> {
|
||||
resetSessionParseCacheForTests()
|
||||
const corpus = await writeSyntheticTranscriptCorpus(options)
|
||||
const store = new SessionSearchStore(join(corpus.root, 'index.sqlite'), (error) => {
|
||||
throw error
|
||||
})
|
||||
const unregister = registerSessionSearchIndexConsumer(store)
|
||||
const stats = createSessionParseStats()
|
||||
for (const path of corpus.files) {
|
||||
await parseAgentSessionFileCached(
|
||||
await sessionCandidate('claude', path),
|
||||
process.platform,
|
||||
stats
|
||||
)
|
||||
}
|
||||
await store.settled()
|
||||
await store.warm()
|
||||
return {
|
||||
corpus,
|
||||
store,
|
||||
release: () => {
|
||||
unregister()
|
||||
resetTranscriptConsumersForTests()
|
||||
resetSessionParseCacheForTests()
|
||||
store.close()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Per-query and overall latency for one scope. */
|
||||
function scopeReport(
|
||||
store: SessionSearchStore,
|
||||
scope: SessionSearchScope
|
||||
): Record<string, unknown> {
|
||||
const engine = new SessionSearchEngine(store)
|
||||
const everything: number[] = []
|
||||
const perQuery: Record<string, Timing & { hits: number; route: string }> = {}
|
||||
for (const { name, request } of QUERIES) {
|
||||
const scoped = { ...request, scope }
|
||||
for (let run = 0; run < WARMUP; run++) {
|
||||
engine.search(scoped)
|
||||
}
|
||||
const samples = Array.from({ length: SAMPLES }, () => time(engine, scoped))
|
||||
everything.push(...samples)
|
||||
const result = engine.search(scoped)
|
||||
perQuery[name] = { ...timing(samples), hits: result.hits.length, route: result.planner.route }
|
||||
}
|
||||
return { ...timing(everything), perQuery }
|
||||
}
|
||||
|
||||
/**
|
||||
* The candidate limit only costs anything once there are more matching sessions
|
||||
* than the limit, so this runs over many short sessions rather than the wide
|
||||
* corpus above. Limits are interleaved sample by sample: run back to back, the
|
||||
* first configuration pays for every page the OS cache had not seen yet and the
|
||||
* ordering alone moves p95 by more than the limit does.
|
||||
*/
|
||||
function candidateSweep(
|
||||
store: SessionSearchStore,
|
||||
limits: readonly number[]
|
||||
): Record<string, unknown> {
|
||||
const request: SessionSearchRequest = { query: 'index', limit: 20 }
|
||||
const engines = new Map(
|
||||
limits.map((limit) => [limit, new SessionSearchEngine(store, { sessionCandidateLimit: limit })])
|
||||
)
|
||||
const samples = new Map(limits.map((limit) => [limit, [] as number[]]))
|
||||
for (let run = 0; run < WARMUP; run++) {
|
||||
for (const engine of engines.values()) {
|
||||
engine.search(request)
|
||||
}
|
||||
}
|
||||
for (let run = 0; run < SAMPLES; run++) {
|
||||
for (const limit of limits) {
|
||||
samples.get(limit)!.push(time(engines.get(limit)!, request))
|
||||
}
|
||||
}
|
||||
const report: Record<string, unknown> = {}
|
||||
for (const limit of limits) {
|
||||
const result = engines.get(limit)!.search(request)
|
||||
report[String(limit)] = {
|
||||
...timing(samples.get(limit)!),
|
||||
truncated: result.truncated.candidates,
|
||||
// Pages a caller could walk before the limit stops handing out sessions.
|
||||
reachablePages: Math.ceil(Math.min(limit, 20 * 1000) / 20)
|
||||
}
|
||||
}
|
||||
return report
|
||||
}
|
||||
|
||||
const wide = await indexCorpus({ sessions: Number(process.env.SESSIONS ?? 40) })
|
||||
let report: string
|
||||
try {
|
||||
const scope = {
|
||||
all: scopeReport(wide.store, 'all'),
|
||||
conversation: scopeReport(wide.store, 'conversation')
|
||||
}
|
||||
wide.release()
|
||||
await rm(wide.corpus.root, { recursive: true, force: true })
|
||||
|
||||
// Many short sessions: what makes the candidate limit binding is the session
|
||||
// count, not the byte count.
|
||||
const many = await indexCorpus({ sessions: 2500, turnsPerSession: 1, seed: 7 })
|
||||
try {
|
||||
report = JSON.stringify(
|
||||
{
|
||||
scopeCorpus: {
|
||||
sessions: wide.corpus.files.length,
|
||||
transcriptMb: Math.round((wide.corpus.transcriptBytes / 1024 / 1024) * 100) / 100,
|
||||
messages: wide.corpus.messageCount
|
||||
},
|
||||
scope,
|
||||
candidateCorpus: {
|
||||
sessions: many.corpus.files.length,
|
||||
transcriptMb: Math.round((many.corpus.transcriptBytes / 1024 / 1024) * 100) / 100
|
||||
},
|
||||
candidateSweep: candidateSweep(many.store, [200, 600, 1200, 2400])
|
||||
},
|
||||
null,
|
||||
2
|
||||
)
|
||||
} finally {
|
||||
many.release()
|
||||
await rm(many.corpus.root, { recursive: true, force: true })
|
||||
}
|
||||
} catch (error) {
|
||||
await rm(wide.corpus.root, { recursive: true, force: true })
|
||||
throw error
|
||||
}
|
||||
|
||||
// Why a file as well as stdout: a runner that intercepts console output
|
||||
// (vitest does) would otherwise swallow the whole report.
|
||||
const out = process.env.BENCH_OUT
|
||||
if (out) {
|
||||
await writeFile(out, `${report}\n`)
|
||||
}
|
||||
console.log(report)
|
||||
@@ -0,0 +1,94 @@
|
||||
# Agent session search: query tuning
|
||||
|
||||
What a search costs, and what the knobs in `src/main/ai-vault-search/session-search-engine.ts`
|
||||
buy. Every number here comes from `config/scripts/session-search-query-benchmark.ts`
|
||||
over the synthetic corpus in `session-search-synthetic-corpus.ts`. Nothing in this
|
||||
file was measured against a real transcript, and the benchmark must never be
|
||||
pointed at one.
|
||||
|
||||
## Running it
|
||||
|
||||
The benchmark is a top-level-await module that imports the main-process tree by
|
||||
extensionless path, so it needs a bundler-backed runner rather than bare `node`:
|
||||
|
||||
```sh
|
||||
cat > src/main/ai-vault-search/zz-bench.test.ts <<'EOF'
|
||||
import { it } from 'vitest'
|
||||
it('runs', { timeout: 600_000 }, async () => {
|
||||
await import('../../../config/scripts/session-search-query-benchmark')
|
||||
})
|
||||
EOF
|
||||
BENCH_OUT=/tmp/ss-query-bench.json pnpm test src/main/ai-vault-search/zz-bench.test.ts
|
||||
rm src/main/ai-vault-search/zz-bench.test.ts
|
||||
```
|
||||
|
||||
`BENCH_OUT` exists because vitest intercepts `console.log`; the report is written
|
||||
to that path as well as printed.
|
||||
|
||||
## Scope: what the second FTS table buys a reader
|
||||
|
||||
Corpus: 40 synthetic Claude transcripts, 10.5 MB, 9,600 messages, indexed through
|
||||
the real store. Eight queries, one per rung of the route ladder plus the two
|
||||
shapes that skip it; 5 warm-up runs and 25 samples each. Apple silicon, warm page
|
||||
cache. Milliseconds.
|
||||
|
||||
| Scope | p50 | p95 |
|
||||
| -------------- | ---- | ----- |
|
||||
| `all` | 8.46 | 41.74 |
|
||||
| `conversation` | 5.32 | 12.20 |
|
||||
|
||||
Per query, `all` then `conversation` (p50 / p95):
|
||||
|
||||
| Query | `all` | `conversation` |
|
||||
| ------------------------------------------------ | ------------- | -------------- |
|
||||
| `"terminal reattach"` (phrase) | 5.03 / 5.86 | 2.77 / 3.78 |
|
||||
| `resolveTerminalPath` (identifier) | 8.47 / 9.54 | 5.39 / 17.89 |
|
||||
| `src/main/…/session-transcript-reader.ts` (path) | 11.96 / 39.31 | 5.55 / 17.27 |
|
||||
| `why is the daemon snapshot stale` (prose) | 10.78 / 25.23 | 7.11 / 12.34 |
|
||||
| `reattahc worktre` (typo repair) | 10.41 / 115.05| 6.73 / 9.41 |
|
||||
| `index` (common term) | 7.25 / 29.99 | 4.98 / 11.51 |
|
||||
| `repo:app-3` (operator only) | 0.10 / 0.60 | 0.10 / 0.20 |
|
||||
| `worktree` scoped to one cwd | 2.10 / 2.60 | 1.48 / 2.06 |
|
||||
|
||||
Reading it:
|
||||
|
||||
- `conversation` is about 1.6x faster at p50 and 3.4x at p95. That gap is the
|
||||
answer to "what is the second table for": it is the corpus a keystroke can
|
||||
afford, and it holds no tool output, so it is also the corpus where a match is
|
||||
something a person wrote.
|
||||
- An operator-only query never touches FTS at all. It is a range seek on
|
||||
`sessions_cwd_key`, and it costs a tenth of a millisecond.
|
||||
- Typo repair's p95 in `all` is the worst number on the page. The repair walks
|
||||
`messages_vocab` per prefix, and the first walk after a cold statement cache
|
||||
pays for the b-tree pages. It is a first-query cost, not a per-query one.
|
||||
|
||||
## `sessionCandidateLimit`
|
||||
|
||||
The reviewer's F13: this is a tunable default, not a constant. It bounds how many
|
||||
sessions the SQL hands ranking, so it bounds both retrieval cost and how deep a
|
||||
caller can page before the answer simply stops.
|
||||
|
||||
The limit only costs anything once more sessions match than the limit allows, so
|
||||
this is measured over a second corpus: 2,500 one-turn transcripts, 10.9 MB, every
|
||||
one of them matching the query. Limits are interleaved sample by sample, because
|
||||
run back to back the first configuration pays for every page the OS cache had not
|
||||
seen and the ordering alone moves p95 further than the limit does.
|
||||
|
||||
| Limit | p50 | p95 | Pages of 20 a caller can reach |
|
||||
| ----- | ----- | ----- | ------------------------------ |
|
||||
| 200 | 9.50 | 13.18 | 10 |
|
||||
| 600 | 10.86 | 20.30 | 30 |
|
||||
| 1200 | 12.65 | 15.20 | 60 |
|
||||
| 2400 | 17.68 | 27.08 | 120 |
|
||||
|
||||
600 is the default: it costs about 14% over 200 at p50 and buys three times the
|
||||
reachable depth, and the curve only turns steep past 1200. A host with a much
|
||||
larger index can raise it; the result's `truncated.candidates` says when the limit
|
||||
was the thing that cut the answer, so a caller never has to guess.
|
||||
|
||||
What is **not** measured here is relevance. These numbers say what a limit costs,
|
||||
not what it retrieves. The MRR figures quoted in the BM25 weights
|
||||
(`session-search-retrieval.ts`) and in the identifier shadow column
|
||||
(`session-search-identifier-split.ts`) come from the original retrieval shoot-out
|
||||
on real transcripts and are not reproducible from this repository. Any change to
|
||||
the limit justified on relevance grounds needs an eval set, not this benchmark.
|
||||
Reference in New Issue
Block a user