From e53dd0cf1da7123061fd839f3028da4fbab5d64d Mon Sep 17 00:00:00 2001 From: Jinwoo-H Date: Thu, 10 Sep 2026 14:20:35 -0400 Subject: [PATCH] test(ai-vault-search): price a per-file commit and the ceiling that bounds it The write benchmark now times every transaction, because with one per file that is the whole stall a file costs. A second phase indexes a single synthetic 100 MB transcript, which is what the commit ceiling is chosen against. --- .../session-search-retention-benchmark.ts | 35 ++++--- .../scripts/session-search-write-benchmark.ts | 97 +++++++++++++++++-- 2 files changed, 112 insertions(+), 20 deletions(-) diff --git a/config/scripts/session-search-retention-benchmark.ts b/config/scripts/session-search-retention-benchmark.ts index 904783a76b7..ac8cf9a69df 100644 --- a/config/scripts/session-search-retention-benchmark.ts +++ b/config/scripts/session-search-retention-benchmark.ts @@ -7,7 +7,7 @@ import { syntheticCandidate, syntheticSession, userMessages -} from '../../src/main/ai-vault-search/session-search-staged-write-test-fixture' +} from '../../src/main/ai-vault-search/session-search-index-test-fixture' import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' import SyncDatabase from '../../src/main/sqlite/sync-database' @@ -29,14 +29,11 @@ async function sampleLoopStalls(running: () => boolean, intervals: number[]): Pr } } -/** What a search would still return: rows whose session is neither staged nor tombstoned. */ +/** What a search would still return: rows whose session row is still there. */ function visibleRows(db: SyncDatabase): number { return ( db - .prepare( - `SELECT count(*) AS n FROM visible_messages m - JOIN visible_sessions s ON s.id = m.session_row_id` - ) + .prepare(`SELECT count(*) AS n FROM messages m JOIN sessions s ON s.id = m.session_row_id`) .get() as { n: number } ).n } @@ -49,18 +46,21 @@ try { const store = new SessionSearchStore(path, (error) => errors.push(error)) let reader: SyncDatabase | null = null try { - const staged = store.beginWrite(syntheticCandidate(), 'replace', 0)! + const write = store.beginWrite(syntheticCandidate(), 'replace', 0)! for (const message of userMessages( 'synthetic benchmark needle repeated context for a representative coding conversation with commands and paths src/example.ts', ROWS )) { - staged.add(message) + write.add(message) } assert.equal( - staged.publish({ session: syntheticSession(), byteOffset: 4096, incomplete: false }), + write.commit({ + session: syntheticSession(), + byteOffset: 4096, + incomplete: false + }), true ) - staged.discard() assert.deepEqual(errors, []) // Truncating first is what makes walBytes below the purge's own growth. const checkpoint = new SyncDatabase(path) @@ -78,7 +78,9 @@ try { const raw = new SyncDatabase(path) try { raw.exec('BEGIN IMMEDIATE') - const ids = raw.prepare('SELECT id FROM messages').all() as { id: number }[] + const ids = raw.prepare('SELECT id FROM messages').all() as { + id: number + }[] for (const { id } of ids) { raw.prepare('DELETE FROM messages_fts WHERE rowid=?').run(id) raw.prepare('DELETE FROM conversation_fts WHERE rowid=?').run(id) @@ -91,8 +93,9 @@ try { } else { let purging = true const purge = store.purgeOlderThan(Date.now() + 60_000) - // Hiding is immediate: the tombstone commits in the first chunk, so a read one - // turn in already sees nothing, long before the rows are gone. + // Hiding is immediate: cutting the session loose from its file is the + // first transaction, so a read one turn in already sees nothing, long + // before the rows are gone. const hiddenEarly = yieldToEventLoop().then(() => visibleRows(probe)) const sampler = sampleLoopStalls(() => purging, intervals) await purge @@ -110,7 +113,11 @@ try { try { for (const table of ['messages_fts', 'conversation_fts']) { assert.equal( - (after.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { n: number }).n, + ( + after.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { + n: number + } + ).n, 0 ) } diff --git a/config/scripts/session-search-write-benchmark.ts b/config/scripts/session-search-write-benchmark.ts index c9c65871c4f..70540209024 100644 --- a/config/scripts/session-search-write-benchmark.ts +++ b/config/scripts/session-search-write-benchmark.ts @@ -16,11 +16,35 @@ import SyncDatabase from '../../src/main/sqlite/sync-database' // The cost model owed to the two-FTS-table decision. Everything runs through the // real transcript reader and SessionSearchStore over a synthetic corpus, so the -// numbers include tokenization, the identifier shadow column, both FTS tables and -// the post-write cleanup a live index pays for. Never point this at a real -// transcript tree. +// numbers include tokenization, the identifier shadow column and both FTS tables. +// Never point this at a real transcript tree. -/** Staging flushes synchronously, so a peer chain samples the gap each read leaves. */ +/** + * How long the longest single transaction held the process. + * + * With one transaction per file that is the whole stall a file costs, so it is + * the number the commit ceiling exists to bound. Measured by wrapping `exec`, + * because the writer's transactions are the only ones this benchmark runs. + */ +function recordTransactionDurations(durations: number[]): () => void { + const exec = SyncDatabase.prototype.exec + let started = 0 + SyncDatabase.prototype.exec = function (this: SyncDatabase, sql: string): void { + if (sql === 'BEGIN IMMEDIATE') { + started = performance.now() + } + exec.call(this, sql) + if (sql === 'COMMIT' && started > 0) { + durations.push(performance.now() - started) + started = 0 + } + } + return () => { + SyncDatabase.prototype.exec = exec + } +} + +/** The writer commits synchronously, so a peer chain samples the gap each read leaves. */ async function sampleLoopStalls(running: () => boolean, stalls: number[]): Promise { let previous = performance.now() while (running()) { @@ -56,6 +80,8 @@ try { const store = new SessionSearchStore(indexPath, (error) => errors.push(error)) const unregister = registerSessionSearchIndexConsumer(store) const stalls: number[] = [] + const transactions: number[] = [] + const restoreExec = recordTransactionDurations(transactions) let indexing = true try { const stats = createSessionParseStats() @@ -70,16 +96,19 @@ try { } indexing = false await sampler + restoreExec() const rebuildMs = performance.now() - started assert.deepEqual(errors, []) const reader = new SyncDatabase(indexPath, { readonly: true }) try { const rows = ( - reader.prepare('SELECT count(*) AS n FROM visible_messages').get() as { n: number } + reader.prepare('SELECT count(*) AS n FROM messages').get() as { + n: number + } ).n const sessions = ( - reader.prepare('SELECT count(*) AS n FROM visible_sessions').get() as { + reader.prepare('SELECT count(*) AS n FROM sessions').get() as { n: number } ).n @@ -89,6 +118,7 @@ try { Math.round((value / (corpus.transcriptBytes / (1024 * 1024))) * 10) / 10 const fileBytes = (await stat(indexPath)).size stalls.sort((a, b) => a - b) + transactions.sort((a, b) => a - b) console.log( JSON.stringify( { @@ -110,6 +140,8 @@ try { }, writeAmplification: Math.round((bytes.total / corpus.transcriptBytes) * 100) / 100, fileWriteAmplification: Math.round((fileBytes / corpus.transcriptBytes) * 100) / 100, + transactions: transactions.length, + maxTransactionMs: Math.round((transactions.at(-1) ?? 0) * 100) / 100, maxLoopStallMs: Math.round(stalls.at(-1) ?? 0), p95LoopStallMs: Math.round(stalls[Math.floor(stalls.length * 0.95)] ?? 0), loopStallSamples: stalls.length, @@ -124,6 +156,7 @@ try { } } finally { indexing = false + restoreExec() unregister() resetTranscriptConsumersForTests() resetSessionParseCacheForTests() @@ -132,3 +165,55 @@ try { } finally { await rm(corpus.root, { recursive: true, force: true }) } + +// Phase two: one transcript far larger than any real one, to price the ceiling +// that decides whether a file commits once or in chunks. +const largeTurns = Number(process.env.ORCA_SEARCH_BENCH_LARGE_TURNS ?? 23_000) +const large = await writeSyntheticTranscriptCorpus({ + sessions: 1, + turnsPerSession: largeTurns, + seed: 2 +}) +const largeIndexPath = join(large.root, 'index.sqlite') +try { + const errors: unknown[] = [] + const store = new SessionSearchStore(largeIndexPath, (error) => errors.push(error)) + const unregister = registerSessionSearchIndexConsumer(store) + const transactions: number[] = [] + const restoreExec = recordTransactionDurations(transactions) + try { + const stats = createSessionParseStats() + const started = performance.now() + await parseAgentSessionFileCached( + await sessionCandidate('claude', large.files[0]!), + process.platform, + stats + ) + const indexMs = performance.now() - started + restoreExec() + assert.deepEqual(errors, []) + transactions.sort((a, b) => a - b) + console.log( + JSON.stringify( + { + phase: 'single-large-file', + transcriptMb: Math.round((large.transcriptBytes / (1024 * 1024)) * 100) / 100, + indexMs: Math.round(indexMs), + transactions: transactions.length, + maxTransactionMs: Math.round(transactions.at(-1) ?? 0), + indexMb: Math.round(((await stat(largeIndexPath)).size / (1024 * 1024)) * 100) / 100 + }, + null, + 2 + ) + ) + } finally { + restoreExec() + unregister() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store.close() + } +} finally { + await rm(large.root, { recursive: true, force: true }) +}