diff --git a/src/main/ai-vault-search/session-search-index-writer.test.ts b/src/main/ai-vault-search/session-search-index-writer.test.ts new file mode 100644 index 00000000000..200e7e15feb --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-writer.test.ts @@ -0,0 +1,126 @@ +import { afterEach, beforeEach, expect, it } from 'vitest' +import { SessionSearchIndexConsumer } from './session-search-index-consumer' +import { + openSessionSearchIndexFile, + syntheticCandidate, + syntheticSession, + SYNTHETIC_TRANSCRIPT, + userMessages, + type SessionSearchIndexFile +} from './session-search-staged-write-test-fixture' +import { SessionSearchStore } from './session-search-store' + +// The store is driven directly here. Every guard below is also shadowed by the +// consumer's own check, so a test that goes through the consumer proves nothing +// about which of the two is holding. + +let index: SessionSearchIndexFile +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + index = await openSessionSearchIndexFile('ss-index-writer') + errors = [] + store = new SessionSearchStore(index.path, (error) => errors.push(error)) +}) + +afterEach(async () => { + store.close() + await index.close() +}) + +function count(table: string): number { + return (index.db.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { n: number }).n +} + +function publishRead(previousByteOffset: number, byteOffset: number, text: string): boolean { + const staged = store.beginWrite( + syntheticCandidate(), + previousByteOffset === 0 ? 'replace' : 'append', + previousByteOffset + ) + if (!staged) { + return false + } + for (const message of userMessages(text, 2)) { + staged.add(message) + } + const published = staged.publish({ session: syntheticSession(), byteOffset, incomplete: false }) + staged.discard() + return published +} + +it('refuses an append whose predecessor offset is not the published cursor', () => { + expect(publishRead(0, 100, 'first')).toBe(true) + + expect(store.beginWrite(syntheticCandidate(), 'append', 900)).toBeNull() + expect(store.beginWrite(syntheticCandidate(), 'append', 99)).toBeNull() + // The one offset that does continue the published span is accepted. + expect(store.beginWrite(syntheticCandidate(), 'append', 100)).not.toBeNull() +}) + +it('refuses to publish a stage whose cursor moved underneath it', async () => { + const candidate = syntheticCandidate() + const stale = store.beginWrite(candidate, 'replace', 0)! + for (const message of userMessages('stalegeneration', 40)) { + stale.add(message) + } + // A second read of the same path finishes first. Without the parse file lane + // this is the overlap that would otherwise resurrect the stale rows. + expect(publishRead(0, 200, 'winninggeneration')).toBe(true) + + expect(stale.publish({ session: syntheticSession(), byteOffset: 100, incomplete: false })).toBe( + false + ) + stale.discard() + await store.purgeOlderThan(null) + + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(200) + expect(count('visible_sessions')).toBe(1) + expect(count('messages')).toBe(2) + expect(count('search_pending_deletes')).toBe(0) + expect(errors).toEqual([]) +}) + +it('refuses to publish a stage whose file was removed mid-read', async () => { + expect(publishRead(0, 100, 'firstgeneration')).toBe(true) + const staged = store.beginWrite(syntheticCandidate(), 'append', 100)! + for (const message of userMessages('afterremoval', 10)) { + staged.add(message) + } + store.removeFile(SYNTHETIC_TRANSCRIPT) + + expect(staged.publish({ session: syntheticSession(), byteOffset: 300, incomplete: false })).toBe( + false + ) + staged.discard() + await store.purgeOlderThan(null) + + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)).toBeNull() + expect(count('sessions')).toBe(0) + expect(count('messages')).toBe(0) +}) + +it('declines a behind cursor in beginRead before it ever reaches the store', () => { + const attempted: number[] = [] + const stub = { + acceptsCandidate: () => true, + indexedFile: () => ({ byteOffset: 100, mtimeMs: 1, sizeBytes: 1 }), + beginWrite: (_candidate: unknown, _mode: unknown, previousByteOffset: number) => { + attempted.push(previousByteOffset) + return { add: () => undefined, publish: () => true, discard: () => undefined } + }, + markStale: () => undefined + } as unknown as SessionSearchStore + const consumer = new SessionSearchIndexConsumer(stub) + + expect( + consumer.beginRead({ candidate: syntheticCandidate(), mode: 'append', previousByteOffset: 900 }) + ).toBeNull() + // The store was never asked, so the writer's own guard cannot be what refused. + expect(attempted).toEqual([]) + expect( + consumer.beginRead({ candidate: syntheticCandidate(), mode: 'append', previousByteOffset: 100 }) + ).not.toBeNull() + expect(attempted).toEqual([100]) +}) diff --git a/src/main/ai-vault-search/session-search-redaction.test.ts b/src/main/ai-vault-search/session-search-redaction.test.ts index a35ce485a88..c41ec3ac981 100644 --- a/src/main/ai-vault-search/session-search-redaction.test.ts +++ b/src/main/ai-vault-search/session-search-redaction.test.ts @@ -1,4 +1,5 @@ import { expect, it } from 'vitest' +import { PROVIDER_PATTERNS } from '../observability/redactor' import { redactSessionSearchText } from './session-search-redaction' const AWS_KEY = 'AKIAIOSFODNN7EXAMPLE' @@ -25,3 +26,37 @@ it('leaves ordinary transcript shapes searchable', () => { expect(redactSessionSearchText(benign)).toBe(benign) } }) + +// Census: SECRET_ANCHOR gates all eight fingerprint passes, so a pattern whose +// shape has no anchor is silently never applied. One fixture per tag, and the +// list must stay exhaustive. +const ANCHORED_FIXTURES: Record = { + 'anthropic-key': `sk-ant-${'a1B2c3D4e5F6g7H8i9J0k1L2m3N4o5P6q7R8s9T0'}`, + 'openai-key': `sk-proj-${'a1B2c3D4e5F6g7H8i9J0k1L2m3N4o5P6'}`, + 'github-token': `ghp_${'A1b2C3d4E5f6G7h8I9j0K1l2M3n4O5p6Q7r8'}`, + 'aws-access-key-id': AWS_KEY, + 'aws-secret-access-key': `aws_secret_access_key = ${'A1b2C3d4E5f6G7h8I9j0K1l2M3n4O5p6Q7r8S9t0'}`, + jwt: JWT, + 'slack-token': `xoxb-${'1234567890-abcdefghij'}`, + pem: '-----BEGIN RSA PRIVATE KEY-----\nMIIEowIBAAKCAQEA\n-----END RSA PRIVATE KEY-----' +} + +it('has an anchored fixture for every provider pattern', () => { + expect(Object.keys(ANCHORED_FIXTURES).sort()).toEqual( + PROVIDER_PATTERNS.map((pattern) => pattern.tag).sort() + ) +}) + +it.each(PROVIDER_PATTERNS.map((pattern) => pattern.tag))( + 'reaches the %s pattern past the anchor gate', + (tag) => { + const fixture = ANCHORED_FIXTURES[tag] + const pattern = PROVIDER_PATTERNS.find((entry) => entry.tag === tag)! + pattern.re.lastIndex = 0 + // The fixture is a real match for its own pattern... + expect(pattern.re.test(fixture)).toBe(true) + pattern.re.lastIndex = 0 + // ...and the anchor lets that pattern run at all. + expect(redactSessionSearchText(fixture)).toContain(`[redacted:${tag}]`) + } +) diff --git a/src/main/ai-vault-search/session-search-visible-read-ratchet.test.ts b/src/main/ai-vault-search/session-search-visible-read-ratchet.test.ts new file mode 100644 index 00000000000..be3a03cc285 --- /dev/null +++ b/src/main/ai-vault-search/session-search-visible-read-ratchet.test.ts @@ -0,0 +1,75 @@ +import { readdir, readFile } from 'node:fs/promises' +import { join } from 'node:path' +import { expect, it } from 'vitest' +import { VISIBLE_MESSAGES, VISIBLE_SESSIONS } from './session-search-schema' + +// Why a ratchet and not a type: publish makes a read's rows visible atomically, +// but it does that by flipping `batch_id` and `index_ready`, not by moving the +// FTS rows — those land in the staging flushes. An FTS table on its own still +// holds staged and tombstoned rows, and only the views subtract them. Every read +// site has to join one, and nothing in SQL can force that. + +const FTS_TABLES = /\b(?:messages_fts|conversation_fts)\b/ +// The view by its resolved name or by the constant a module interpolates. +const VISIBLE_VIEW = new RegExp( + `\\b(?:${VISIBLE_MESSAGES}|${VISIBLE_SESSIONS}|VISIBLE_MESSAGES|VISIBLE_SESSIONS)\\b` +) +const STRING_LITERAL = /`(?:[^`\\]|\\[\s\S])*`|'(?:[^'\\\n]|\\.)*'|"(?:[^"\\\n]|\\.)*"/g + +/** SQL literals in `source` that read an FTS table without subtracting staged rows. */ +export function unguardedFtsReads(source: string): string[] { + const offenders: string[] = [] + for (const [literal] of source.matchAll(STRING_LITERAL)) { + if (!/\bSELECT\b/i.test(literal) || !FTS_TABLES.test(literal)) { + continue + } + if (!VISIBLE_VIEW.test(literal)) { + offenders.push(literal.replaceAll(/\s+/g, ' ').slice(0, 120)) + } + } + return offenders +} + +async function productionSources(): Promise<{ name: string; text: string }[]> { + const dir = import.meta.dirname + const names = (await readdir(dir)).filter( + (name) => + name.endsWith('.ts') && !name.endsWith('.test.ts') && !name.endsWith('-test-fixture.ts') + ) + return Promise.all( + names.map(async (name) => ({ name, text: await readFile(join(dir, name), 'utf-8') })) + ) +} + +it('flags a read of an FTS table that does not subtract staged rows', () => { + const bare = String.raw`db.prepare('SELECT rowid FROM messages_fts WHERE messages_fts MATCH ?')` + expect(unguardedFtsReads(bare)).toHaveLength(1) + + const joined = String.raw`db.prepare(\`SELECT s.id FROM conversation_fts + JOIN visible_messages m ON m.id = conversation_fts.rowid + WHERE conversation_fts MATCH ?\`)` + expect(unguardedFtsReads(joined)).toEqual([]) + + // A delete is not a read: retention removes staged rows on purpose. + expect( + unguardedFtsReads(String.raw`db.prepare('DELETE FROM messages_fts WHERE rowid = ?')`) + ).toEqual([]) + // Two statements in one file must not blur into one match. + expect( + unguardedFtsReads( + String.raw`db.prepare('SELECT path FROM files'); db.prepare('INSERT INTO messages_fts(rowid) VALUES (?)')` + ) + ).toEqual([]) +}) + +it('reads every FTS table through a visibility view across the index modules', async () => { + const sources = await productionSources() + expect(sources.length).toBeGreaterThan(10) + // Pointed at the real corpus, so a query module is covered the moment it lands. + expect(sources.some((file) => FTS_TABLES.test(file.text))).toBe(true) + + const offenders = sources.flatMap((file) => + unguardedFtsReads(file.text).map((statement) => `${file.name}: ${statement}`) + ) + expect(offenders).toEqual([]) +})