mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 08:02:43 +00:00
test(ai-vault-search): price a per-file commit and the ceiling that bounds it
The write benchmark now times every transaction, because with one per file that is the whole stall a file costs. A second phase indexes a single synthetic 100 MB transcript, which is what the commit ceiling is chosen against.
This commit is contained in:
@@ -7,7 +7,7 @@ import {
|
||||
syntheticCandidate,
|
||||
syntheticSession,
|
||||
userMessages
|
||||
} from '../../src/main/ai-vault-search/session-search-staged-write-test-fixture'
|
||||
} from '../../src/main/ai-vault-search/session-search-index-test-fixture'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import SyncDatabase from '../../src/main/sqlite/sync-database'
|
||||
|
||||
@@ -29,14 +29,11 @@ async function sampleLoopStalls(running: () => boolean, intervals: number[]): Pr
|
||||
}
|
||||
}
|
||||
|
||||
/** What a search would still return: rows whose session is neither staged nor tombstoned. */
|
||||
/** What a search would still return: rows whose session row is still there. */
|
||||
function visibleRows(db: SyncDatabase): number {
|
||||
return (
|
||||
db
|
||||
.prepare(
|
||||
`SELECT count(*) AS n FROM visible_messages m
|
||||
JOIN visible_sessions s ON s.id = m.session_row_id`
|
||||
)
|
||||
.prepare(`SELECT count(*) AS n FROM messages m JOIN sessions s ON s.id = m.session_row_id`)
|
||||
.get() as { n: number }
|
||||
).n
|
||||
}
|
||||
@@ -49,18 +46,21 @@ try {
|
||||
const store = new SessionSearchStore(path, (error) => errors.push(error))
|
||||
let reader: SyncDatabase | null = null
|
||||
try {
|
||||
const staged = store.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const write = store.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
for (const message of userMessages(
|
||||
'synthetic benchmark needle repeated context for a representative coding conversation with commands and paths src/example.ts',
|
||||
ROWS
|
||||
)) {
|
||||
staged.add(message)
|
||||
write.add(message)
|
||||
}
|
||||
assert.equal(
|
||||
staged.publish({ session: syntheticSession(), byteOffset: 4096, incomplete: false }),
|
||||
write.commit({
|
||||
session: syntheticSession(),
|
||||
byteOffset: 4096,
|
||||
incomplete: false
|
||||
}),
|
||||
true
|
||||
)
|
||||
staged.discard()
|
||||
assert.deepEqual(errors, [])
|
||||
// Truncating first is what makes walBytes below the purge's own growth.
|
||||
const checkpoint = new SyncDatabase(path)
|
||||
@@ -78,7 +78,9 @@ try {
|
||||
const raw = new SyncDatabase(path)
|
||||
try {
|
||||
raw.exec('BEGIN IMMEDIATE')
|
||||
const ids = raw.prepare('SELECT id FROM messages').all() as { id: number }[]
|
||||
const ids = raw.prepare('SELECT id FROM messages').all() as {
|
||||
id: number
|
||||
}[]
|
||||
for (const { id } of ids) {
|
||||
raw.prepare('DELETE FROM messages_fts WHERE rowid=?').run(id)
|
||||
raw.prepare('DELETE FROM conversation_fts WHERE rowid=?').run(id)
|
||||
@@ -91,8 +93,9 @@ try {
|
||||
} else {
|
||||
let purging = true
|
||||
const purge = store.purgeOlderThan(Date.now() + 60_000)
|
||||
// Hiding is immediate: the tombstone commits in the first chunk, so a read one
|
||||
// turn in already sees nothing, long before the rows are gone.
|
||||
// Hiding is immediate: cutting the session loose from its file is the
|
||||
// first transaction, so a read one turn in already sees nothing, long
|
||||
// before the rows are gone.
|
||||
const hiddenEarly = yieldToEventLoop().then(() => visibleRows(probe))
|
||||
const sampler = sampleLoopStalls(() => purging, intervals)
|
||||
await purge
|
||||
@@ -110,7 +113,11 @@ try {
|
||||
try {
|
||||
for (const table of ['messages_fts', 'conversation_fts']) {
|
||||
assert.equal(
|
||||
(after.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { n: number }).n,
|
||||
(
|
||||
after.prepare(`SELECT count(*) AS n FROM ${table}`).get() as {
|
||||
n: number
|
||||
}
|
||||
).n,
|
||||
0
|
||||
)
|
||||
}
|
||||
|
||||
@@ -16,11 +16,35 @@ import SyncDatabase from '../../src/main/sqlite/sync-database'
|
||||
|
||||
// The cost model owed to the two-FTS-table decision. Everything runs through the
|
||||
// real transcript reader and SessionSearchStore over a synthetic corpus, so the
|
||||
// numbers include tokenization, the identifier shadow column, both FTS tables and
|
||||
// the post-write cleanup a live index pays for. Never point this at a real
|
||||
// transcript tree.
|
||||
// numbers include tokenization, the identifier shadow column and both FTS tables.
|
||||
// Never point this at a real transcript tree.
|
||||
|
||||
/** Staging flushes synchronously, so a peer chain samples the gap each read leaves. */
|
||||
/**
|
||||
* How long the longest single transaction held the process.
|
||||
*
|
||||
* With one transaction per file that is the whole stall a file costs, so it is
|
||||
* the number the commit ceiling exists to bound. Measured by wrapping `exec`,
|
||||
* because the writer's transactions are the only ones this benchmark runs.
|
||||
*/
|
||||
function recordTransactionDurations(durations: number[]): () => void {
|
||||
const exec = SyncDatabase.prototype.exec
|
||||
let started = 0
|
||||
SyncDatabase.prototype.exec = function (this: SyncDatabase, sql: string): void {
|
||||
if (sql === 'BEGIN IMMEDIATE') {
|
||||
started = performance.now()
|
||||
}
|
||||
exec.call(this, sql)
|
||||
if (sql === 'COMMIT' && started > 0) {
|
||||
durations.push(performance.now() - started)
|
||||
started = 0
|
||||
}
|
||||
}
|
||||
return () => {
|
||||
SyncDatabase.prototype.exec = exec
|
||||
}
|
||||
}
|
||||
|
||||
/** The writer commits synchronously, so a peer chain samples the gap each read leaves. */
|
||||
async function sampleLoopStalls(running: () => boolean, stalls: number[]): Promise<void> {
|
||||
let previous = performance.now()
|
||||
while (running()) {
|
||||
@@ -56,6 +80,8 @@ try {
|
||||
const store = new SessionSearchStore(indexPath, (error) => errors.push(error))
|
||||
const unregister = registerSessionSearchIndexConsumer(store)
|
||||
const stalls: number[] = []
|
||||
const transactions: number[] = []
|
||||
const restoreExec = recordTransactionDurations(transactions)
|
||||
let indexing = true
|
||||
try {
|
||||
const stats = createSessionParseStats()
|
||||
@@ -70,16 +96,19 @@ try {
|
||||
}
|
||||
indexing = false
|
||||
await sampler
|
||||
restoreExec()
|
||||
const rebuildMs = performance.now() - started
|
||||
assert.deepEqual(errors, [])
|
||||
|
||||
const reader = new SyncDatabase(indexPath, { readonly: true })
|
||||
try {
|
||||
const rows = (
|
||||
reader.prepare('SELECT count(*) AS n FROM visible_messages').get() as { n: number }
|
||||
reader.prepare('SELECT count(*) AS n FROM messages').get() as {
|
||||
n: number
|
||||
}
|
||||
).n
|
||||
const sessions = (
|
||||
reader.prepare('SELECT count(*) AS n FROM visible_sessions').get() as {
|
||||
reader.prepare('SELECT count(*) AS n FROM sessions').get() as {
|
||||
n: number
|
||||
}
|
||||
).n
|
||||
@@ -89,6 +118,7 @@ try {
|
||||
Math.round((value / (corpus.transcriptBytes / (1024 * 1024))) * 10) / 10
|
||||
const fileBytes = (await stat(indexPath)).size
|
||||
stalls.sort((a, b) => a - b)
|
||||
transactions.sort((a, b) => a - b)
|
||||
console.log(
|
||||
JSON.stringify(
|
||||
{
|
||||
@@ -110,6 +140,8 @@ try {
|
||||
},
|
||||
writeAmplification: Math.round((bytes.total / corpus.transcriptBytes) * 100) / 100,
|
||||
fileWriteAmplification: Math.round((fileBytes / corpus.transcriptBytes) * 100) / 100,
|
||||
transactions: transactions.length,
|
||||
maxTransactionMs: Math.round((transactions.at(-1) ?? 0) * 100) / 100,
|
||||
maxLoopStallMs: Math.round(stalls.at(-1) ?? 0),
|
||||
p95LoopStallMs: Math.round(stalls[Math.floor(stalls.length * 0.95)] ?? 0),
|
||||
loopStallSamples: stalls.length,
|
||||
@@ -124,6 +156,7 @@ try {
|
||||
}
|
||||
} finally {
|
||||
indexing = false
|
||||
restoreExec()
|
||||
unregister()
|
||||
resetTranscriptConsumersForTests()
|
||||
resetSessionParseCacheForTests()
|
||||
@@ -132,3 +165,55 @@ try {
|
||||
} finally {
|
||||
await rm(corpus.root, { recursive: true, force: true })
|
||||
}
|
||||
|
||||
// Phase two: one transcript far larger than any real one, to price the ceiling
|
||||
// that decides whether a file commits once or in chunks.
|
||||
const largeTurns = Number(process.env.ORCA_SEARCH_BENCH_LARGE_TURNS ?? 23_000)
|
||||
const large = await writeSyntheticTranscriptCorpus({
|
||||
sessions: 1,
|
||||
turnsPerSession: largeTurns,
|
||||
seed: 2
|
||||
})
|
||||
const largeIndexPath = join(large.root, 'index.sqlite')
|
||||
try {
|
||||
const errors: unknown[] = []
|
||||
const store = new SessionSearchStore(largeIndexPath, (error) => errors.push(error))
|
||||
const unregister = registerSessionSearchIndexConsumer(store)
|
||||
const transactions: number[] = []
|
||||
const restoreExec = recordTransactionDurations(transactions)
|
||||
try {
|
||||
const stats = createSessionParseStats()
|
||||
const started = performance.now()
|
||||
await parseAgentSessionFileCached(
|
||||
await sessionCandidate('claude', large.files[0]!),
|
||||
process.platform,
|
||||
stats
|
||||
)
|
||||
const indexMs = performance.now() - started
|
||||
restoreExec()
|
||||
assert.deepEqual(errors, [])
|
||||
transactions.sort((a, b) => a - b)
|
||||
console.log(
|
||||
JSON.stringify(
|
||||
{
|
||||
phase: 'single-large-file',
|
||||
transcriptMb: Math.round((large.transcriptBytes / (1024 * 1024)) * 100) / 100,
|
||||
indexMs: Math.round(indexMs),
|
||||
transactions: transactions.length,
|
||||
maxTransactionMs: Math.round(transactions.at(-1) ?? 0),
|
||||
indexMb: Math.round(((await stat(largeIndexPath)).size / (1024 * 1024)) * 100) / 100
|
||||
},
|
||||
null,
|
||||
2
|
||||
)
|
||||
)
|
||||
} finally {
|
||||
restoreExec()
|
||||
unregister()
|
||||
resetTranscriptConsumersForTests()
|
||||
resetSessionParseCacheForTests()
|
||||
store.close()
|
||||
}
|
||||
} finally {
|
||||
await rm(large.root, { recursive: true, force: true })
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user