fix(ai-vault-search): never chunk a read that cannot name its session

`updateProvisionalSession` is fed by the resumable readers alone. Claude and
Codex carry an `identity()` off their fold; `readWholeTranscript` supplies
none, because a format rewritten in place has no resumable state to ask. So a
Grok, Cursor, Gemini or OpenCode file large enough to pass the commit ceiling
published its chunks under a session with an empty id, an empty title and a
null cwd — rows that answer searches at once and that an interrupted read
leaves behind for good, since the re-read that heals them is the same read
starting over.

`add` now commits a chunk only while the read can name what it is writing. A
read with no identity, or one whose parser has decoded no id yet, keeps
buffering and commits whole at `finish`. The whole-file formats are the ones
with nothing to give and they are small — the largest on this machine is 5 MB
— so buffering one to the end costs nothing, and chunking stays reserved for
the readers that can say which session a prefix belongs to.

`updateProvisionalSession` takes a non-null identity now, so the invariant is
the type rather than a guard that silently wrote nothing.
This commit is contained in:
Jinwoo-H
2026-09-10 16:58:21 -04:00
parent ca5e16e636
commit 72cc341c4e
3 changed files with 108 additions and 35 deletions
@@ -45,14 +45,12 @@ export class SessionSearchFileRecords {
*
* Rows a chunk commits answer searches the moment they land, so the session
* they hang off has to be nameable before the read producing it ends — and it
* may never end, because a crash between chunks leaves exactly this row. The
* final commit overwrites all of it from the decoded session; until then the
* title in particular is provisional.
* may never end, because a crash between chunks leaves exactly this row. That
* is why the identity is required rather than optional: a read that has none
* does not chunk at all. The final commit overwrites all of it from the
* decoded session; until then the title in particular is provisional.
*/
updateProvisionalSession(rowId: number, identity: TranscriptSessionIdentity | null): void {
if (!identity) {
return
}
updateProvisionalSession(rowId: number, identity: TranscriptSessionIdentity): void {
this.db
.prepare(
`UPDATE sessions SET session_id = ?, title = ?, cwd = ?, cwd_key = ?,
@@ -201,10 +201,22 @@ it('shows a reader on another handle one generation or the other, never a mixtur
// Four of these fill the 400-char ceiling the two tests below construct.
const CHUNKED_MESSAGE = `chunkedneedle ${'filler '.repeat(12)}nd`
const PROVISIONAL_IDENTITY = {
sessionId: 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee',
cwd: '/repo/app',
title: 'provisional title',
createdAt: '2026-05-01T10:00:00.000Z',
updatedAt: '2026-05-01T10:05:00.000Z'
}
// Only a read that can name its session chunks at all, so every test below that
// wants a chunk has to supply one.
const named = (): typeof PROVISIONAL_IDENTITY => PROVISIONAL_IDENTITY
it('leaves the session consistent after every chunk of a file too large for one transaction', () => {
expect(CHUNKED_MESSAGE.length).toBe(100)
const writer = new SessionSearchIndexWriter(index.db, 400)
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
for (const [position, message] of userMessages(CHUNKED_MESSAGE, 10).entries()) {
write.add(message)
const rows = counts(index.db).messages
@@ -238,7 +250,7 @@ it('leaves the session consistent after every chunk of a file too large for one
it('holds the ceiling against a single message larger than it', () => {
const writer = new SessionSearchIndexWriter(index.db, 8000)
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
const exec = SyncDatabase.prototype.exec
let opened = 0
vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function (
@@ -269,17 +281,9 @@ it('holds the ceiling against a single message larger than it', () => {
expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 3 })
})
const PROVISIONAL_IDENTITY = {
sessionId: 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee',
cwd: '/repo/app',
title: 'provisional title',
createdAt: '2026-05-01T10:00:00.000Z',
updatedAt: '2026-05-01T10:05:00.000Z'
}
it('names a session on its first chunk, not only when the read ends', () => {
const writer = new SessionSearchIndexWriter(index.db, 400)
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => PROVISIONAL_IDENTITY)!
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
write.add(message)
}
@@ -312,24 +316,66 @@ it('names a session on its first chunk, not only when the read ends', () => {
})
})
it('leaves a chunked session unnamed only while the parser has no id yet', () => {
it('commits a whole-file read over the ceiling in one transaction, never a chunk', () => {
// The whole-file readers (Grok, Cursor, Gemini, OpenCode) pass no identity:
// their formats are rewritten in place and have no resumable state to ask.
const writer = new SessionSearchIndexWriter(index.db, 400)
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => null)!
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
const exec = SyncDatabase.prototype.exec
let opened = 0
vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function (
this: SyncDatabase,
sql: string
) {
if (sql === 'BEGIN IMMEDIATE') {
opened += 1
}
exec.call(this, sql)
})
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
write.add(message)
// Chunking here would publish rows under a session with an empty id, an
// empty title and a null cwd, and an interrupted read would leave that
// prefix answering searches for good.
expect(counts(index.db)).toMatchObject({ sessions: 0, messages: 0, files: 0 })
}
expect(write.commit({ session: syntheticSession(), byteOffset: 4096, incomplete: false })).toBe(
true
)
vi.restoreAllMocks()
// A parser with nothing decoded is not a reason to write a wrong id; the row
// is still created, and the next chunk fills it in.
expect(opened).toBe(1)
expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 10, full: 10 })
// And a real cursor, not the partial sentinel a chunk would have left.
expect(writer.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(4096)
})
it('starts chunking only once the parser has an id to name the session with', () => {
const writer = new SessionSearchIndexWriter(index.db, 400)
let decoded: typeof PROVISIONAL_IDENTITY | null = null
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => decoded)!
for (const message of userMessages(CHUNKED_MESSAGE, 4)) {
write.add(message)
}
// Past the ceiling, but the parser has decoded nothing: the buffer keeps
// growing rather than naming a session it cannot name.
expect(counts(index.db).messages).toBe(0)
decoded = PROVISIONAL_IDENTITY
write.add(userMessages(CHUNKED_MESSAGE, 1)[0]!)
// Everything held goes with the first chunk that can say what it is.
expect(counts(index.db).messages).toBe(5)
expect(index.db.prepare('SELECT session_id, cwd FROM sessions').get()).toEqual({
session_id: '',
cwd: null
session_id: PROVISIONAL_IDENTITY.sessionId,
cwd: '/repo/app'
})
})
it('reports a chunk-partial file as held, and as one that must be read whole', () => {
const writer = new SessionSearchIndexWriter(index.db, 400)
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
write.add(message)
}
@@ -356,7 +402,7 @@ it('reports a chunk-partial file as held, and as one that must be read whole', (
it('re-reads a chunked file whole when its writer died between chunks', () => {
const writer = new SessionSearchIndexWriter(index.db, 400)
const abandoned = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
const abandoned = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
abandoned.add(message)
}
@@ -380,7 +426,7 @@ it('re-reads a chunked file whole when its writer died between chunks', () => {
it('stops a chunked read whose file was removed between its chunks', () => {
const writer = new SessionSearchIndexWriter(index.db, 400)
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
const messages = userMessages(CHUNKED_MESSAGE, 10)
for (const message of messages.slice(0, 4)) {
write.add(message)
@@ -26,7 +26,8 @@ import {
* single commit near a second and the WAL it produces near 64 MB, and it is far
* above the largest real transcript (the 40-session benchmark corpus is 10.5 MB
* in total), so an ordinary file never reaches it. Above the ceiling the read is
* cut into chunks that each leave the index consistent — see `chunked`.
* cut into chunks that each leave the index consistent — but only a read that
* can name its session chunks at all. See `add`.
*/
export const SESSION_SEARCH_COMMIT_CHARS = 32 * 1024 * 1024
@@ -54,7 +55,20 @@ type FileRow = {
type FileCursor = Pick<FileRow, 'session_row_id' | 'byte_offset'>
export type SessionSearchFileWrite = {
/** Buffers one message, committing a chunk when the buffer reaches the ceiling. */
/**
* Buffers one message, committing a chunk when the buffer reaches the ceiling
* — and only while this read can name the session it is writing.
*
* A chunk's rows answer searches the moment they land, so a read with no
* `identity` would publish them under a session with an empty id, an empty
* title and a null cwd, and an interrupted read would leave that prefix
* behind for good. The readers that supply no identity are the whole-file
* ones (Grok, Cursor, Gemini, OpenCode), whose formats are rewritten in place
* and have no resumable state to ask; they are also small — the largest on
* the author's machine is 5 MB — so buffering one to the end and committing
* it whole costs nothing. Chunking stays reserved for the readers that can
* say which session this is before the read ends.
*/
add(message: TranscriptMessage): void
/**
* Writes this file's rows, its session and its cursor in one transaction.
@@ -217,8 +231,14 @@ export class SessionSearchIndexWriter {
)
}
/** `outcome` is null for a chunk of a read that has not reached the file's end. */
const write = (outcome: TranscriptReadOutcome | null): boolean => {
/**
* `outcome` is null for a chunk of a read that has not reached the file's
* end, and `named` is what that chunk writes onto its session row.
*/
const write = (
outcome: TranscriptReadOutcome | null,
named: TranscriptSessionIdentity | null
): boolean => {
const decoded = outcome?.session ?? null
db.exec('BEGIN IMMEDIATE')
try {
@@ -244,11 +264,12 @@ export class SessionSearchIndexWriter {
}
if (decoded) {
this.records.updateSession(decoded, session, hash)
} else {
} else if (named) {
// A chunk's rows answer searches as soon as they land, so the
// session they hang off is written with whatever the parser has
// decoded rather than left empty until a read that may never end.
this.records.updateProvisionalSession(session, identity?.() ?? null)
// `add` refuses to chunk without this, so it is never absent here.
this.records.updateProvisionalSession(session, named)
}
this.records.upsertFile(
candidate,
@@ -283,7 +304,15 @@ export class SessionSearchIndexWriter {
for (const row of searchMessageRows([message])) {
buffer.push(row)
bufferedChars += row.text.length
if (bufferedChars >= this.commitChars && !write(null)) {
if (bufferedChars < this.commitChars) {
continue
}
// Publishing a chunk under a session nothing can identify is worse
// than holding the buffer: the rows answer searches at once, and an
// interrupted read leaves that prefix for good. A read with nothing
// to name it keeps buffering and commits whole at `finish`.
const named = identity?.() ?? null
if (named && !write(null, named)) {
fenced = true
buffer.length = 0
bufferedChars = 0
@@ -291,7 +320,7 @@ export class SessionSearchIndexWriter {
}
}
},
commit: (outcome) => !fenced && write(outcome)
commit: (outcome) => !fenced && write(outcome, null)
}
}