diff --git a/src/main/ai-vault-search/session-search-redaction.test.ts b/src/main/ai-vault-search/session-search-redaction.test.ts new file mode 100644 index 00000000000..90e61e758b0 --- /dev/null +++ b/src/main/ai-vault-search/session-search-redaction.test.ts @@ -0,0 +1,115 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture' +import { redactSessionSearchText } from './session-search-redaction' +import { SessionSearchStore } from './session-search-store' +import { + CLAUDE_SESSION_ID as SESSION_ID, + parseTranscript, + userRecord +} from './session-search-transcript-fixtures' + +const GITHUB_TOKEN = `ghp_${'A1b2C3d4E5f6G7h8I9j0K1l2M3n4O5p6Q7r8'}` +const AWS_KEY = 'AKIAIOSFODNN7EXAMPLE' +const JWT = + 'eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w5N_XgL0n3I9PlFUP0THsR8U' +const PEM = [ + '-----BEGIN RSA PRIVATE KEY-----', + 'MIIEowIBAAKCAQEAqhVmVvXTPQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA', + '-----END RSA PRIVATE KEY-----' +].join('\n') + +let tempRoots: string[] = [] +let store: SessionSearchStore + +beforeEach(async () => { + resetSessionParseCacheForTests() + const root = await makeTempDir() + store = new SessionSearchStore(join(root, 'index.sqlite'), (error) => { + throw error + }) + registerSessionSearchIndexSink(store) +}) + +afterEach(async () => { + registerSessionSearchIndexSink(null) + store.close() + await Promise.all(tempRoots.map((root) => rm(root, { recursive: true, force: true }))) + tempRoots = [] +}) + +async function makeTempDir(): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-session-redaction-')) + tempRoots.push(root) + return root +} + +describe('session search redaction', () => { + it('keeps the fingerprint patterns and leaves ordinary transcript shapes alone', () => { + expect(redactSessionSearchText(`key ${AWS_KEY} here`)).toBe( + 'key [redacted:aws-access-key-id] here' + ) + expect(redactSessionSearchText(`Authorization: Bearer ${JWT}`)).toBe( + 'Authorization: Bearer [redacted:jwt]' + ) + // Opaque (non-JWT) bearer values are covered too; the label survives. + expect(redactSessionSearchText('Bearer abcdefghijklmnopqrstuvwxyz012345')).toBe( + 'Bearer [redacted:bearer-token]' + ) + // Why these and not `redactString`: env-shaped code lines and `token:` prose + // are ordinary transcript content and must stay searchable. + for (const benign of [ + 'MAX_RETRIES = 3', + 'the auth token: refreshed on 401', + 'API_KEY_HEADER' + ]) { + expect(redactSessionSearchText(benign)).toBe(benign) + } + }) + + it('never indexes a credential that appeared in tool output', async () => { + const root = await makeTempDir() + const path = join(root, `${SESSION_ID}.jsonl`) + await writeFile( + path, + `${[ + userRecord(0, 'deploy the staging worker'), + userRecord(1, [ + { + type: 'tool_result', + tool_use_id: 'toolu_1', + content: [ + 'writing deployment credentials to the staging environment', + `github_token=${GITHUB_TOKEN}`, + `aws_access_key_id=${AWS_KEY}`, + `authorization: Bearer ${JWT}`, + PEM, + 'deployment finished with a green rollout' + ].join('\n') + } + ]) + ].join('\n')}\n` + ) + await parseTranscript(path) + + // The surrounding words stay searchable... + const hit = store.search({ query: 'deployment credentials staging' }) + expect(hit.hits).toHaveLength(1) + expect(store.search({ query: 'green rollout' }).hits).toHaveLength(1) + // ...and the redaction marker rides along in the snippet. + expect(store.search({ query: 'redacted' }).hits[0]?.evidence.snippet).toContain('redacted') + + // ...but none of the secrets are findable. + for (const secret of [ + GITHUB_TOKEN, + AWS_KEY, + JWT, + 'MIIEowIBAAKCAQEAqhVmVvXTPQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA' + ]) { + expect(store.search({ query: secret }).hits).toHaveLength(0) + } + }) +}) diff --git a/src/main/ai-vault-search/session-search-redaction.ts b/src/main/ai-vault-search/session-search-redaction.ts new file mode 100644 index 00000000000..13c800a12c7 --- /dev/null +++ b/src/main/ai-vault-search/session-search-redaction.ts @@ -0,0 +1,26 @@ +import { PROVIDER_PATTERNS } from '../observability/redactor' + +// Why not `redactString`: its labeled-kv and .env-line rules are tuned for +// stack traces and are far too eager over transcript text — every pasted +// `MAX_RETRIES = 3` diff line would lose its value, and prose like +// `token: the next token` would lose the word `token` itself, taking the +// searchable content with it. The provider fingerprints are shape-matched and +// safe over prose, so those are reused verbatim, plus the one shape they miss: +// an opaque (non-JWT) bearer token. +const BEARER_TOKEN = /\b(Bearer)\s+[A-Za-z0-9._~+/=-]{16,}/g + +// One scan decides whether the eight fingerprint passes run at all: every +// pattern above is anchored on one of these, so text without them cannot match. +const SECRET_ANCHOR = /sk-|gh[pousr]_|AKIA|eyJ|xox|-----|[Bb]earer|aws_secret_access_key/ + +/** Strips credential-shaped spans so the index (and every snippet) never holds one. */ +export function redactSessionSearchText(text: string): string { + if (!SECRET_ANCHOR.test(text)) { + return text + } + let out = text + for (const { tag, re } of PROVIDER_PATTERNS) { + out = out.replace(re, `[redacted:${tag}]`) + } + return out.replace(BEARER_TOKEN, '$1 [redacted:bearer-token]') +} diff --git a/src/main/ai-vault-search/session-search-schema.ts b/src/main/ai-vault-search/session-search-schema.ts index f15fc7009f0..3e6fd41e04b 100644 --- a/src/main/ai-vault-search/session-search-schema.ts +++ b/src/main/ai-vault-search/session-search-schema.ts @@ -1,7 +1,7 @@ import SyncDatabase from '../sqlite/sync-database' // Bump to drop and rebuild: the index is a cache over the transcripts, never a source. -export const SESSION_SEARCH_SCHEMA_VERSION = 3 +export const SESSION_SEARCH_SCHEMA_VERSION = 4 // unicode61 keeps `_ . - /` inside tokens so paths and identifiers match exactly; // the `identifiers` column carries the split form (see session-search-identifier-split). diff --git a/src/main/ai-vault/session-search-content.ts b/src/main/ai-vault/session-search-content.ts index 50cad8da623..b49f9b9c475 100644 --- a/src/main/ai-vault/session-search-content.ts +++ b/src/main/ai-vault/session-search-content.ts @@ -1,3 +1,4 @@ +import { redactSessionSearchText } from '../ai-vault-search/session-search-redaction' import { captureSessionSearchMessage, isSessionSearchCaptureActive, @@ -168,7 +169,11 @@ export function captureIndexableText( return } const limit = indexed === 'tool' ? SESSION_SEARCH_TOOL_OUTPUT_CAP : SESSION_SEARCH_TEXT_CAP - const cleaned = cap(stripHiddenBlocks(cap(text, limit * 4)), limit).trim() + // Redact before the final cap so a credential straddling it cannot survive in halves. + const cleaned = cap( + redactSessionSearchText(stripHiddenBlocks(cap(text, limit * 4))), + limit + ).trim() if (!cleaned) { return } diff --git a/src/main/observability/redactor.ts b/src/main/observability/redactor.ts index 7ddab42a0c9..8f8b7eedc56 100644 --- a/src/main/observability/redactor.ts +++ b/src/main/observability/redactor.ts @@ -14,7 +14,7 @@ const LABELED_KV = /\b(?:api[-_]?key|token|secret|password|bearer|authorization)\b\s*[:=]\s*(?:Bearer\s+\S+|Token\s+\S+|\S+)/gi // Tagged tokens let triage see what was redacted without the key. Order is most-specific-first: `sk-ant-` before `sk-`, or the Anthropic tag is lost. -const PROVIDER_PATTERNS: { tag: string; re: RegExp }[] = [ +export const PROVIDER_PATTERNS: { tag: string; re: RegExp }[] = [ { tag: 'anthropic-key', re: /sk-ant-[a-zA-Z0-9_-]{40,}/g }, { tag: 'openai-key', re: /sk-(?:proj-)?[a-zA-Z0-9_-]{32,}/g }, { tag: 'github-token', re: /gh[pousr]_[A-Za-z0-9]{36,}/g },