mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 16:02:45 +00:00
fix(ai-vault-search): never chunk a read that cannot name its session
`updateProvisionalSession` is fed by the resumable readers alone. Claude and Codex carry an `identity()` off their fold; `readWholeTranscript` supplies none, because a format rewritten in place has no resumable state to ask. So a Grok, Cursor, Gemini or OpenCode file large enough to pass the commit ceiling published its chunks under a session with an empty id, an empty title and a null cwd — rows that answer searches at once and that an interrupted read leaves behind for good, since the re-read that heals them is the same read starting over. `add` now commits a chunk only while the read can name what it is writing. A read with no identity, or one whose parser has decoded no id yet, keeps buffering and commits whole at `finish`. The whole-file formats are the ones with nothing to give and they are small — the largest on this machine is 5 MB — so buffering one to the end costs nothing, and chunking stays reserved for the readers that can say which session a prefix belongs to. `updateProvisionalSession` takes a non-null identity now, so the invariant is the type rather than a guard that silently wrote nothing.
This commit is contained in:
@@ -45,14 +45,12 @@ export class SessionSearchFileRecords {
|
||||
*
|
||||
* Rows a chunk commits answer searches the moment they land, so the session
|
||||
* they hang off has to be nameable before the read producing it ends — and it
|
||||
* may never end, because a crash between chunks leaves exactly this row. The
|
||||
* final commit overwrites all of it from the decoded session; until then the
|
||||
* title in particular is provisional.
|
||||
* may never end, because a crash between chunks leaves exactly this row. That
|
||||
* is why the identity is required rather than optional: a read that has none
|
||||
* does not chunk at all. The final commit overwrites all of it from the
|
||||
* decoded session; until then the title in particular is provisional.
|
||||
*/
|
||||
updateProvisionalSession(rowId: number, identity: TranscriptSessionIdentity | null): void {
|
||||
if (!identity) {
|
||||
return
|
||||
}
|
||||
updateProvisionalSession(rowId: number, identity: TranscriptSessionIdentity): void {
|
||||
this.db
|
||||
.prepare(
|
||||
`UPDATE sessions SET session_id = ?, title = ?, cwd = ?, cwd_key = ?,
|
||||
|
||||
@@ -201,10 +201,22 @@ it('shows a reader on another handle one generation or the other, never a mixtur
|
||||
// Four of these fill the 400-char ceiling the two tests below construct.
|
||||
const CHUNKED_MESSAGE = `chunkedneedle ${'filler '.repeat(12)}nd`
|
||||
|
||||
const PROVISIONAL_IDENTITY = {
|
||||
sessionId: 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee',
|
||||
cwd: '/repo/app',
|
||||
title: 'provisional title',
|
||||
createdAt: '2026-05-01T10:00:00.000Z',
|
||||
updatedAt: '2026-05-01T10:05:00.000Z'
|
||||
}
|
||||
|
||||
// Only a read that can name its session chunks at all, so every test below that
|
||||
// wants a chunk has to supply one.
|
||||
const named = (): typeof PROVISIONAL_IDENTITY => PROVISIONAL_IDENTITY
|
||||
|
||||
it('leaves the session consistent after every chunk of a file too large for one transaction', () => {
|
||||
expect(CHUNKED_MESSAGE.length).toBe(100)
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
|
||||
for (const [position, message] of userMessages(CHUNKED_MESSAGE, 10).entries()) {
|
||||
write.add(message)
|
||||
const rows = counts(index.db).messages
|
||||
@@ -238,7 +250,7 @@ it('leaves the session consistent after every chunk of a file too large for one
|
||||
|
||||
it('holds the ceiling against a single message larger than it', () => {
|
||||
const writer = new SessionSearchIndexWriter(index.db, 8000)
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
|
||||
const exec = SyncDatabase.prototype.exec
|
||||
let opened = 0
|
||||
vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function (
|
||||
@@ -269,17 +281,9 @@ it('holds the ceiling against a single message larger than it', () => {
|
||||
expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 3 })
|
||||
})
|
||||
|
||||
const PROVISIONAL_IDENTITY = {
|
||||
sessionId: 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee',
|
||||
cwd: '/repo/app',
|
||||
title: 'provisional title',
|
||||
createdAt: '2026-05-01T10:00:00.000Z',
|
||||
updatedAt: '2026-05-01T10:05:00.000Z'
|
||||
}
|
||||
|
||||
it('names a session on its first chunk, not only when the read ends', () => {
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => PROVISIONAL_IDENTITY)!
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
|
||||
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
|
||||
write.add(message)
|
||||
}
|
||||
@@ -312,24 +316,66 @@ it('names a session on its first chunk, not only when the read ends', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('leaves a chunked session unnamed only while the parser has no id yet', () => {
|
||||
it('commits a whole-file read over the ceiling in one transaction, never a chunk', () => {
|
||||
// The whole-file readers (Grok, Cursor, Gemini, OpenCode) pass no identity:
|
||||
// their formats are rewritten in place and have no resumable state to ask.
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => null)!
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const exec = SyncDatabase.prototype.exec
|
||||
let opened = 0
|
||||
vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function (
|
||||
this: SyncDatabase,
|
||||
sql: string
|
||||
) {
|
||||
if (sql === 'BEGIN IMMEDIATE') {
|
||||
opened += 1
|
||||
}
|
||||
exec.call(this, sql)
|
||||
})
|
||||
|
||||
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
|
||||
write.add(message)
|
||||
// Chunking here would publish rows under a session with an empty id, an
|
||||
// empty title and a null cwd, and an interrupted read would leave that
|
||||
// prefix answering searches for good.
|
||||
expect(counts(index.db)).toMatchObject({ sessions: 0, messages: 0, files: 0 })
|
||||
}
|
||||
expect(write.commit({ session: syntheticSession(), byteOffset: 4096, incomplete: false })).toBe(
|
||||
true
|
||||
)
|
||||
vi.restoreAllMocks()
|
||||
|
||||
// A parser with nothing decoded is not a reason to write a wrong id; the row
|
||||
// is still created, and the next chunk fills it in.
|
||||
expect(opened).toBe(1)
|
||||
expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 10, full: 10 })
|
||||
// And a real cursor, not the partial sentinel a chunk would have left.
|
||||
expect(writer.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(4096)
|
||||
})
|
||||
|
||||
it('starts chunking only once the parser has an id to name the session with', () => {
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
let decoded: typeof PROVISIONAL_IDENTITY | null = null
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => decoded)!
|
||||
for (const message of userMessages(CHUNKED_MESSAGE, 4)) {
|
||||
write.add(message)
|
||||
}
|
||||
// Past the ceiling, but the parser has decoded nothing: the buffer keeps
|
||||
// growing rather than naming a session it cannot name.
|
||||
expect(counts(index.db).messages).toBe(0)
|
||||
|
||||
decoded = PROVISIONAL_IDENTITY
|
||||
write.add(userMessages(CHUNKED_MESSAGE, 1)[0]!)
|
||||
|
||||
// Everything held goes with the first chunk that can say what it is.
|
||||
expect(counts(index.db).messages).toBe(5)
|
||||
expect(index.db.prepare('SELECT session_id, cwd FROM sessions').get()).toEqual({
|
||||
session_id: '',
|
||||
cwd: null
|
||||
session_id: PROVISIONAL_IDENTITY.sessionId,
|
||||
cwd: '/repo/app'
|
||||
})
|
||||
})
|
||||
|
||||
it('reports a chunk-partial file as held, and as one that must be read whole', () => {
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
|
||||
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
|
||||
write.add(message)
|
||||
}
|
||||
@@ -356,7 +402,7 @@ it('reports a chunk-partial file as held, and as one that must be read whole', (
|
||||
|
||||
it('re-reads a chunked file whole when its writer died between chunks', () => {
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
const abandoned = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const abandoned = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
|
||||
for (const message of userMessages(CHUNKED_MESSAGE, 10)) {
|
||||
abandoned.add(message)
|
||||
}
|
||||
@@ -380,7 +426,7 @@ it('re-reads a chunked file whole when its writer died between chunks', () => {
|
||||
|
||||
it('stops a chunked read whose file was removed between its chunks', () => {
|
||||
const writer = new SessionSearchIndexWriter(index.db, 400)
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)!
|
||||
const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)!
|
||||
const messages = userMessages(CHUNKED_MESSAGE, 10)
|
||||
for (const message of messages.slice(0, 4)) {
|
||||
write.add(message)
|
||||
|
||||
@@ -26,7 +26,8 @@ import {
|
||||
* single commit near a second and the WAL it produces near 64 MB, and it is far
|
||||
* above the largest real transcript (the 40-session benchmark corpus is 10.5 MB
|
||||
* in total), so an ordinary file never reaches it. Above the ceiling the read is
|
||||
* cut into chunks that each leave the index consistent — see `chunked`.
|
||||
* cut into chunks that each leave the index consistent — but only a read that
|
||||
* can name its session chunks at all. See `add`.
|
||||
*/
|
||||
export const SESSION_SEARCH_COMMIT_CHARS = 32 * 1024 * 1024
|
||||
|
||||
@@ -54,7 +55,20 @@ type FileRow = {
|
||||
type FileCursor = Pick<FileRow, 'session_row_id' | 'byte_offset'>
|
||||
|
||||
export type SessionSearchFileWrite = {
|
||||
/** Buffers one message, committing a chunk when the buffer reaches the ceiling. */
|
||||
/**
|
||||
* Buffers one message, committing a chunk when the buffer reaches the ceiling
|
||||
* — and only while this read can name the session it is writing.
|
||||
*
|
||||
* A chunk's rows answer searches the moment they land, so a read with no
|
||||
* `identity` would publish them under a session with an empty id, an empty
|
||||
* title and a null cwd, and an interrupted read would leave that prefix
|
||||
* behind for good. The readers that supply no identity are the whole-file
|
||||
* ones (Grok, Cursor, Gemini, OpenCode), whose formats are rewritten in place
|
||||
* and have no resumable state to ask; they are also small — the largest on
|
||||
* the author's machine is 5 MB — so buffering one to the end and committing
|
||||
* it whole costs nothing. Chunking stays reserved for the readers that can
|
||||
* say which session this is before the read ends.
|
||||
*/
|
||||
add(message: TranscriptMessage): void
|
||||
/**
|
||||
* Writes this file's rows, its session and its cursor in one transaction.
|
||||
@@ -217,8 +231,14 @@ export class SessionSearchIndexWriter {
|
||||
)
|
||||
}
|
||||
|
||||
/** `outcome` is null for a chunk of a read that has not reached the file's end. */
|
||||
const write = (outcome: TranscriptReadOutcome | null): boolean => {
|
||||
/**
|
||||
* `outcome` is null for a chunk of a read that has not reached the file's
|
||||
* end, and `named` is what that chunk writes onto its session row.
|
||||
*/
|
||||
const write = (
|
||||
outcome: TranscriptReadOutcome | null,
|
||||
named: TranscriptSessionIdentity | null
|
||||
): boolean => {
|
||||
const decoded = outcome?.session ?? null
|
||||
db.exec('BEGIN IMMEDIATE')
|
||||
try {
|
||||
@@ -244,11 +264,12 @@ export class SessionSearchIndexWriter {
|
||||
}
|
||||
if (decoded) {
|
||||
this.records.updateSession(decoded, session, hash)
|
||||
} else {
|
||||
} else if (named) {
|
||||
// A chunk's rows answer searches as soon as they land, so the
|
||||
// session they hang off is written with whatever the parser has
|
||||
// decoded rather than left empty until a read that may never end.
|
||||
this.records.updateProvisionalSession(session, identity?.() ?? null)
|
||||
// `add` refuses to chunk without this, so it is never absent here.
|
||||
this.records.updateProvisionalSession(session, named)
|
||||
}
|
||||
this.records.upsertFile(
|
||||
candidate,
|
||||
@@ -283,7 +304,15 @@ export class SessionSearchIndexWriter {
|
||||
for (const row of searchMessageRows([message])) {
|
||||
buffer.push(row)
|
||||
bufferedChars += row.text.length
|
||||
if (bufferedChars >= this.commitChars && !write(null)) {
|
||||
if (bufferedChars < this.commitChars) {
|
||||
continue
|
||||
}
|
||||
// Publishing a chunk under a session nothing can identify is worse
|
||||
// than holding the buffer: the rows answer searches at once, and an
|
||||
// interrupted read leaves that prefix for good. A read with nothing
|
||||
// to name it keeps buffering and commits whole at `finish`.
|
||||
const named = identity?.() ?? null
|
||||
if (named && !write(null, named)) {
|
||||
fenced = true
|
||||
buffer.length = 0
|
||||
bufferedChars = 0
|
||||
@@ -291,7 +320,7 @@ export class SessionSearchIndexWriter {
|
||||
}
|
||||
}
|
||||
},
|
||||
commit: (outcome) => !fenced && write(outcome)
|
||||
commit: (outcome) => !fenced && write(outcome, null)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user