fix(ai-vault-search): continue the cursor of a file that decoded no session

A read that decodes no session — an excluded Codex worker transcript — advances
the cursor and leaves `files.session_row_id` null. Every later append was then
declined, so a file that only ever grows would be re-read whole on every pass
for the rest of its life.

An append now only has to continue this index's own cursor; it creates the
session row when there is none to resume. That also collapses the `append` flag,
which meant both "resume this session" and "do not own it", into the resumed row
id itself.
This commit is contained in:
Jinwoo-H
2026-09-10 13:15:09 -04:00
parent 4774fd212d
commit ed0f1f7ac0
2 changed files with 39 additions and 11 deletions
@@ -62,6 +62,31 @@ it('appends onto its own cursor and carries the content hash forward', async ()
expect(store.takeStale()).toEqual([])
})
it('appends onto a file it read through and decoded no session from', async () => {
// An excluded Codex worker transcript: read through, nothing to index, and
// still growing. Its cursor is sound, so a re-read of the whole file every
// pass buys nothing.
replayTranscriptRead({
messages: userMessages('excluded span', 3),
outcome: { session: null, byteOffset: 100 }
})
await store.settled()
expect(cursor()).toBe(100)
expect(store.takeStale()).toEqual([])
replayTranscriptRead({
mode: 'append',
previousByteOffset: 100,
messages: userMessages('decoded at last', 2),
outcome: { byteOffset: 220 }
})
await store.settled()
expect(visibleMessages()).toBe(2)
expect(cursor()).toBe(220)
expect(store.takeStale()).toEqual([])
})
it('declines an append that starts past its own cursor and records the file', async () => {
replayTranscriptRead({ messages: userMessages('indexed span', 3), outcome: { byteOffset: 100 } })
await store.settled()
@@ -110,14 +110,14 @@ export class SessionSearchIndexWriter {
): SessionSearchStagedWrite | null {
const path = candidate.file.path
const existing = this.file(path)
const append =
mode === 'append' &&
existing?.session_row_id != null &&
existing.byte_offset === previousByteOffset
if (mode === 'append' && !append) {
if (mode === 'append' && existing?.byte_offset !== previousByteOffset) {
return null
}
return this.stage(candidate, existing, append)
// A file the index read through and decoded no session from still has a
// cursor worth continuing: it has no session row to hang new rows off, so
// this read makes one. Declining instead would force a whole re-read of
// that file on every pass for as long as it grows.
return this.stage(candidate, existing, mode === 'append' ? existing.session_row_id : null)
}
/** Invalidation hides the generation immediately; cleanup does the expensive deletes later. */
@@ -146,19 +146,20 @@ export class SessionSearchIndexWriter {
.get(path) as ExistingFile | undefined
}
/** `resumed` is the session row this read continues, or null when it starts one. */
private stage(
candidate: SessionFileCandidate,
existing: ExistingFile | undefined,
append: boolean
resumed: number | null
): SessionSearchStagedWrite {
const db = this.db
const path = candidate.file.path
let hash = append ? this.records.contentHash(existing!.session_row_id!) : EMPTY_CONTENT_HASH
let hash = resumed === null ? EMPTY_CONTENT_HASH : this.records.contentHash(resumed)
let sessionId: number
let batchId: number
db.exec('BEGIN IMMEDIATE')
try {
sessionId = append ? existing!.session_row_id! : this.records.createStagingSession(candidate)
sessionId = resumed ?? this.records.createStagingSession(candidate)
batchId = Number(
db.prepare('INSERT INTO search_write_batches(session_row_id) VALUES (?)').run(sessionId)
.lastInsertRowid
@@ -241,7 +242,7 @@ export class SessionSearchIndexWriter {
try {
if (outcome.session) {
this.records.updateSession(outcome.session, sessionId, hash)
if (!append && existing?.session_row_id != null) {
if (resumed === null && existing?.session_row_id != null) {
retireSearchSession(db, existing.session_row_id)
}
db.prepare('UPDATE sessions SET index_ready=1 WHERE id=?').run(sessionId)
@@ -272,7 +273,9 @@ export class SessionSearchIndexWriter {
// A surviving batch row means publish never made these rows visible,
// whatever ended the stage.
if (db.prepare('SELECT 1 FROM search_write_batches WHERE id=?').get(batchId)) {
discardSearchBatch(db, sessionId, batchId, !append)
// Owning the session means this read created it, so retiring it takes
// the whole staging generation with it.
discardSearchBatch(db, sessionId, batchId, resumed === null)
}
}
}