From e0281999db3af85b830389fde23f554cb0e7962d Mon Sep 17 00:00:00 2001 From: Jinwoo-H Date: Mon, 7 Sep 2026 22:12:56 -0400 Subject: [PATCH] refactor(ai-vault-search): consolidate the review fixes into one owner per rule Second pass over the session-search delta. Keeps the staged-write, streaming capture, consent-gate and SQL-side filtering designs; removes the layers that had accumulated around them. Store and schema (schema 10): row visibility is two SQL views instead of four hand-written predicates; publish nulls messages.batch_id and drops the batch row, so the messages view is `batch_id IS NULL` and a recycled batch id can no longer hide published rows; `published` column and its index removed; WAL checkpoint guard taken off every read path and sampled on the write loop only; maintenance class split into three plain functions; cwd_key indexed and the scope filter switched to a range seek with an EXPLAIN plan assertion. Service and capture: one write shape (streamingCapture flag and the dead array branch deleted); configure applies policy to the store once; one shutdown path (dispose removed) with an ownership guard so a late close cannot unregister a replacement service's sink; unverifiable sources are returned with a caveat instead of filtered like deletions; a transient write failure no longer pins the indexing badge at error; refresh lane reuses stableInFlightKey; message channel handles concurrent checkpoints. Query: relevance sort now groups by session before the candidate limit (the must-fix was only applied to newest); one operator parser shared by panel and backend with the apostrophe bug fixed; OR within a key, AND across keys; one documented case rule; received enums tolerate unknown values; result type derived from the schema; projection applied on the desktop IPC path; tokenizer contract pinned against fts5vocab. Remote and CLI: SSH and local ids rejected at the RPC boundary for all four methods; method-name regex sniffing removed from both transports; CommandSpec gained booleanFlags/repeatableFlags so args.ts carries no command vocabulary; settings update no longer blocks or fails on scanner reconfiguration; relay owner keeps its lease through caller cancellation and answers index-status on an unreadable policy; one SESSION_SEARCH_METHODS record feeds every caller. Renderer: coverage polling stops once the index settles and re-arms on focus; search results feed the coverage store instead of re-fetching; the `updating` state and header line are deleted in favour of the list's own loading state; local-only notice only when a remote host is in scope; status-bar segment is read-only and opens settings; ownerKey folded into the args key. Providers: OpenCode capture resets (not clears) the parse deadline; capture and preview reads degrade to a scan issue instead of dropping the session; a consumer write failure is not counted as a worker death; response union discriminated on `kind`; worker host split out of the client. Every behavioural fix carries a test that fails when the fix is reverted. --- .../session-search-retention-benchmark.ts | 34 +-- .../scripts/session-search-write-benchmark.ts | 30 ++- src/cli/agent-session-search-format.ts | 5 +- src/cli/args.ts | 66 +++-- src/cli/command-scoped-flag-parsing.test.ts | 44 ++++ src/cli/command-spec.ts | 5 + src/cli/handlers/search.ts | 24 +- src/cli/index.ts | 5 +- src/cli/runtime/transport.test.ts | 73 +++--- src/cli/runtime/transport.ts | 22 -- src/cli/search-command-arguments.test.ts | 43 ++-- src/cli/session-search-all-hosts.ts | 36 ++- src/cli/session-search-host-query.ts | 64 +++-- src/cli/specs/search.ts | 10 + .../session-search-cancel-publication.test.ts | 6 +- ...session-search-capture-regressions.test.ts | 36 ++- .../session-search-consent.test.ts | 74 +++++- .../session-search-durable-reuse.test.ts | 44 ++-- .../session-search-enablement.test.ts | 66 +++++ .../session-search-enablement.ts | 25 ++ .../session-search-fts5-contract.test.ts | 43 ++++ .../session-search-hit-ranking.ts | 139 +++++++++++ .../session-search-index-compaction.ts | 22 ++ .../session-search-index-writer.ts | 64 +++-- .../session-search-indexing-progress.test.ts | 18 ++ .../session-search-indexing-progress.ts | 8 +- .../session-search-maintenance.ts | 81 ------- ...ssion-search-opencode-cancellation.test.ts | 2 +- .../session-search-opencode-freshness.test.ts | 20 ++ .../session-search-page-warmup.ts | 27 +++ .../session-search-parse-candidates.ts | 14 +- .../session-search-pause.test.ts | 6 +- ...y.ts => session-search-pending-deletes.ts} | 15 +- .../session-search-query-log.ts | 24 ++ .../session-search-query-operators.test.ts | 25 ++ .../session-search-query-planner.ts | 13 +- .../session-search-query-regressions.test.ts | 202 ++++++++++------ .../ai-vault-search/session-search-query.ts | 149 +++--------- .../session-search-refresh-lane.ts | 3 +- .../session-search-retention-delete.test.ts | 68 +++--- .../session-search-retention-delete.ts | 2 - .../session-search-row-filter.ts | 111 +++++---- .../session-search-schema.test.ts | 68 +++++- .../ai-vault-search/session-search-schema.ts | 28 ++- .../ai-vault-search/session-search-service.ts | 97 ++++---- .../session-search-session-row.ts | 33 --- .../session-search-source-presence.test.ts | 5 +- .../session-search-source-presence.ts | 7 +- .../session-search-source-refill.test.ts | 19 +- .../session-search-staged-write-fixtures.ts | 42 ---- ...ession-search-staged-write-test-fixture.ts | 87 +++++++ .../session-search-staged-write.test.ts | 146 +++++++----- .../session-search-store.test.ts | 22 +- .../ai-vault-search/session-search-store.ts | 65 +++-- .../session-search-streaming-write.test.ts | 31 ++- .../session-search-typo-policy.test.ts | 12 +- .../session-search-typo-repair.ts | 67 +++--- .../session-search-wal-budget.test.ts | 50 ++-- src/main/ai-vault/session-newest-files.ts | 6 +- .../session-scanner-directory-reader.test.ts | 15 +- .../ai-vault/session-scanner-discovery.ts | 73 +++--- .../ai-vault/session-scanner-jsonl-reader.ts | 5 +- ...ession-scanner-opencode-pending-request.ts | 40 ++++ .../session-scanner-opencode-sqlite-schema.ts | 29 +++ ...nner-opencode-sqlite-worker-client.test.ts | 223 ++++++++++++----- ...n-scanner-opencode-sqlite-worker-client.ts | 149 ++++-------- ...anner-opencode-sqlite-worker-entry.test.ts | 36 ++- ...on-scanner-opencode-sqlite-worker-entry.ts | 12 +- ...scanner-opencode-sqlite-worker-protocol.ts | 25 +- .../session-scanner-opencode-sqlite.test.ts | 43 ++++ .../session-scanner-opencode-sqlite.ts | 68 +++--- .../session-scanner-opencode-worker-host.ts | 108 +++++++++ .../ai-vault/session-scanner-parse-cache.ts | 10 +- ...-scanner-service-entry-search-init.test.ts | 53 +++++ .../ai-vault/session-scanner-service-entry.ts | 9 +- .../ai-vault/session-scanner-service-env.ts | 1 - src/main/ai-vault/session-search-capture.ts | 19 +- .../ai-vault/session-search-indexed-parse.ts | 12 - .../session-search-message-channel.test.ts | 40 ++++ .../session-search-message-channel.ts | 19 +- ...ssion-search-opencode-cancellation.test.ts | 13 +- ...session-search-opencode-capture-channel.ts | 75 ++++++ .../session-search-opencode-content.ts | 75 ++++-- .../session-search-opencode-worker-capture.ts | 2 +- ...session-search-opencode-worker-receiver.ts | 94 -------- src/main/ipc/ai-vault.ts | 6 +- src/main/ipc/settings.ts | 8 +- .../rpc/methods/ai-vault-search.test.ts | 42 +++- src/main/runtime/rpc/methods/ai-vault.ts | 20 +- src/main/runtime/runtime-ai-vault-commands.ts | 28 ++- ...runtime-ai-vault-search-durability.test.ts | 57 ++++- src/relay/ai-vault-handler.test.ts | 15 +- src/relay/ai-vault-handler.ts | 18 +- src/relay/ai-vault-service-client-state.ts | 9 +- src/relay/ai-vault-service-client.ts | 4 +- src/relay/ai-vault-service-entry.ts | 17 +- src/relay/ai-vault-service-protocol.test.ts | 43 ++++ src/relay/ai-vault-service-protocol.ts | 26 +- src/relay/session-search-owner-policy-file.ts | 74 ++++++ .../session-search-owner-progress.test.ts | 6 +- src/relay/session-search-owner.test.ts | 54 +++++ src/relay/session-search-owner.ts | 140 +++++------ .../components/right-sidebar/AiVaultPanel.tsx | 10 +- .../right-sidebar/AiVaultPanelHeader.tsx | 10 +- .../AiVaultSessionVirtualList.tsx | 3 +- .../right-sidebar/AiVaultVirtualRow.tsx | 10 +- .../right-sidebar/ai-vault-host-scope.ts | 9 + .../ai-vault-search-coverage-poll.test.ts | 225 +++++++++++++++--- .../ai-vault-search-coverage-poll.ts | 9 +- .../ai-vault-search-coverage-store.ts | 103 ++++++-- .../ai-vault-session-launch-actions.test.tsx | 133 +++++++++++ .../ai-vault-session-launch-actions.ts | 42 ++-- ...-vault-session-resume-in-chat-workspace.ts | 75 +++--- .../ai-vault-session-resume-in-chat.test.ts | 78 +++--- .../ai-vault-session-resume-in-chat.ts | 41 +--- .../ai-vault-session-search-request.test.ts | 2 - .../ai-vault-session-search-request.ts | 3 - .../ai-vault-session-search-results.test.tsx | 31 ++- .../ai-vault-session-search-results.ts | 36 +-- .../settings/AgentSessionHistoryPane.tsx | 11 +- .../SessionSearchIndexingPanel.test.tsx | 15 +- .../settings/SessionSearchIndexingPanel.tsx | 61 +++-- .../SessionSearchStatusSegment.test.tsx | 103 ++++++++ .../status-bar/SessionSearchStatusSegment.tsx | 84 +++---- src/renderer/src/components/ui/progress.tsx | 3 +- .../src/i18n/en-runtime-required.json | 4 +- src/renderer/src/i18n/locales/en.json | 4 +- src/shared/ai-vault-search-contract.ts | 73 +++--- src/shared/ai-vault-search-coverage.ts | 14 +- src/shared/ai-vault-search-projection.test.ts | 54 ++++- src/shared/ai-vault-search-projection.ts | 38 ++- .../ai-vault-search-query-operators.test.ts | 28 ++- src/shared/ai-vault-search-query-operators.ts | 93 +++++--- src/shared/ai-vault-search-rpc-methods.ts | 29 +++ src/shared/ai-vault-search-settings.ts | 8 - src/shared/ai-vault-session-filters.ts | 59 ++--- src/shared/cli-argument-boundary.ts | 35 ++- src/shared/protocol-version.ts | 2 - src/shared/remote-runtime-request-socket.ts | 5 +- 139 files changed, 3948 insertions(+), 1963 deletions(-) create mode 100644 src/cli/command-scoped-flag-parsing.test.ts create mode 100644 src/main/ai-vault-search/session-search-hit-ranking.ts create mode 100644 src/main/ai-vault-search/session-search-index-compaction.ts delete mode 100644 src/main/ai-vault-search/session-search-maintenance.ts create mode 100644 src/main/ai-vault-search/session-search-page-warmup.ts rename src/main/ai-vault-search/{session-search-write-recovery.ts => session-search-pending-deletes.ts} (63%) create mode 100644 src/main/ai-vault-search/session-search-query-log.ts delete mode 100644 src/main/ai-vault-search/session-search-session-row.ts delete mode 100644 src/main/ai-vault-search/session-search-staged-write-fixtures.ts create mode 100644 src/main/ai-vault-search/session-search-staged-write-test-fixture.ts create mode 100644 src/main/ai-vault/session-scanner-opencode-pending-request.ts create mode 100644 src/main/ai-vault/session-scanner-opencode-sqlite-schema.ts create mode 100644 src/main/ai-vault/session-scanner-opencode-worker-host.ts create mode 100644 src/main/ai-vault/session-scanner-service-entry-search-init.test.ts create mode 100644 src/main/ai-vault/session-search-message-channel.test.ts create mode 100644 src/main/ai-vault/session-search-opencode-capture-channel.ts delete mode 100644 src/main/ai-vault/session-search-opencode-worker-receiver.ts create mode 100644 src/relay/ai-vault-service-protocol.test.ts create mode 100644 src/relay/session-search-owner-policy-file.ts create mode 100644 src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.test.tsx create mode 100644 src/renderer/src/components/status-bar/SessionSearchStatusSegment.test.tsx create mode 100644 src/shared/ai-vault-search-rpc-methods.ts diff --git a/config/scripts/session-search-retention-benchmark.ts b/config/scripts/session-search-retention-benchmark.ts index 50a26effb85..5f9e224bc2c 100644 --- a/config/scripts/session-search-retention-benchmark.ts +++ b/config/scripts/session-search-retention-benchmark.ts @@ -4,19 +4,19 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { setImmediate as yieldToEventLoop } from 'node:timers/promises' import SyncDatabase from '../../src/main/sqlite/sync-database' -import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' +import { SessionSearchQuery } from '../../src/main/ai-vault-search/session-search-query' +import { openSessionSearchDatabase } from '../../src/main/ai-vault-search/session-search-schema' import { deleteExpiredSearchFiles } from '../../src/main/ai-vault-search/session-search-retention-delete' -import { SearchWalBackpressureError } from '../../src/main/ai-vault-search/session-search-wal-budget' // Bundle with esbuild --bundle --platform=node, then run on the host under test. const root = await mkdtemp(join(tmpdir(), 'orca-search-retention-bench-')) try { for (const mode of ['whole-file', 'batched', 'batched-pinned-reader']) { const path = join(root, `${mode}.sqlite`) - const store = new SessionSearchStore(path) + const db = openSessionSearchDatabase(path) let reader: SyncDatabase | null = null try { - const db = store.db + const query = new SessionSearchQuery(db) db.exec(`INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,resume_command) VALUES (1,'claude','1','fixture','synthetic benchmark','/fixture','/fixture',''); INSERT INTO files(path,byte_offset,mtime_ms,session_row_id) VALUES ('fixture',1,1,1); @@ -28,7 +28,7 @@ try { FROM messages; INSERT INTO conversation_fts(rowid,user_text) SELECT rowid,user_text FROM messages_fts; COMMIT; PRAGMA wal_checkpoint(TRUNCATE)`) - assert.equal(store.search({ query: 'needle' }).hits.length, 1) + assert.equal(query.execute({ query: 'needle' }, null).hits.length, 1) if (mode === 'batched-pinned-reader') { reader = new SyncDatabase(path, { readonly: true }) reader.exec('BEGIN') @@ -56,29 +56,11 @@ try { () => {}, async () => { intervals.push(performance.now() - previous) - assert.equal(store.search({ query: 'needle' }).hits.length, 0) + assert.equal(query.execute({ query: 'needle' }, null).hits.length, 0) await yieldToEventLoop() previous = performance.now() } - ).catch(async (error: unknown) => { - if (!(error instanceof SearchWalBackpressureError) || !reader) { - throw error - } - console.log( - JSON.stringify({ - mode, - backpressured: true, - walBytes: (await stat(`${path}-wal`)).size - }) - ) - reader.exec('COMMIT') - await deleteExpiredSearchFiles( - db, - null, - () => false, - () => {} - ) - }) + ) } const wallMs = performance.now() - started assert.equal( @@ -106,7 +88,7 @@ try { ) } finally { reader?.close() - store.close() + db.close() } } } finally { diff --git a/config/scripts/session-search-write-benchmark.ts b/config/scripts/session-search-write-benchmark.ts index 00b73f0b953..9495fd5ec30 100644 --- a/config/scripts/session-search-write-benchmark.ts +++ b/config/scripts/session-search-write-benchmark.ts @@ -3,16 +3,19 @@ import { mkdtemp, rm, stat } from 'node:fs/promises' import { join } from 'node:path' import { tmpdir } from 'node:os' import { setImmediate as yieldToEventLoop } from 'node:timers/promises' -import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' +import { deleteExpiredSearchFiles } from '../../src/main/ai-vault-search/session-search-retention-delete' import { SessionSearchIndexWriter } from '../../src/main/ai-vault-search/session-search-index-writer' -import { stagedWriteUpdate } from '../../src/main/ai-vault-search/session-search-staged-write-fixtures' +import { SessionSearchQuery } from '../../src/main/ai-vault-search/session-search-query' +import { openSessionSearchDatabase } from '../../src/main/ai-vault-search/session-search-schema' +import { stagedWriteUpdate } from '../../src/main/ai-vault-search/session-search-staged-write-test-fixture' const root = await mkdtemp(join(tmpdir(), 'orca-search-write-bench-')) try { const path = join(root, 'index.sqlite') - const store = new SessionSearchStore(path) + const db = openSessionSearchDatabase(path) try { - const writer = new SessionSearchIndexWriter(store.db) + const writer = new SessionSearchIndexWriter(db) + const query = new SessionSearchQuery(db) for (const mode of ['replace', 'append', 'replace'] as const) { const update = stagedWriteUpdate( `benchmarkneedle ${'synthetic coding context src/example.ts '.repeat(5)}`, @@ -22,20 +25,18 @@ try { const steps: number[] = [] let before = performance.now() const start = before - await writer.apply( - update, - () => true, - async () => { + await writer.apply(update, { + yieldStep: async () => { steps.push(performance.now() - before) await yieldToEventLoop() before = performance.now() } - ) + }) const wallMs = performance.now() - start if (!steps.length) { steps.push(wallMs) } - assert.equal(store.search({ query: 'benchmarkneedle' }).hits.length, 1) + assert.equal(query.execute({ query: 'benchmarkneedle' }, null).hits.length, 1) console.log( JSON.stringify({ platform: process.platform, @@ -48,10 +49,15 @@ try { walBytes: (await stat(`${path}-wal`)).size }) ) - await store.purgeOlderThan(null) + await deleteExpiredSearchFiles( + db, + null, + () => false, + () => {} + ) } } finally { - store.close() + db.close() } } finally { await rm(root, { recursive: true, force: true }) diff --git a/src/cli/agent-session-search-format.ts b/src/cli/agent-session-search-format.ts index 72caa0c8a55..40ab3193ef7 100644 --- a/src/cli/agent-session-search-format.ts +++ b/src/cli/agent-session-search-format.ts @@ -5,6 +5,7 @@ import { import type { AiVaultSearchIndexStatus } from '../shared/ai-vault-search-settings' import { aiVaultAgentLabel } from '../shared/ai-vault-types' import { aiVaultSearchUnindexedProviders } from '../shared/ai-vault-search-coverage' +import { getRuntimePathBasename } from '../shared/cross-platform-path' import type { AiVaultSearchHit, AiVaultSearchResult } from '../shared/ai-vault-search-types' const ROLE_LABEL: Record = { @@ -40,7 +41,9 @@ function relativeAge(iso: string | null, now = Date.now()): string { } function projectLabel(hit: AiVaultSearchHit): string { - const cwd = hit.cwd ? (hit.cwd.replaceAll('\\', '/').split('/').findLast(Boolean) ?? '—') : '—' + // Why: the path comes from the execution host, so node:path would read a + // Windows path with POSIX rules (and the reverse) when the two disagree. + const cwd = (hit.cwd ? getRuntimePathBasename(hit.cwd) : '') || '—' return hit.branch ? `${cwd} · ${hit.branch}` : cwd } diff --git a/src/cli/args.ts b/src/cli/args.ts index e81a8a47dd6..97bf6e730af 100644 --- a/src/cli/args.ts +++ b/src/cli/args.ts @@ -5,7 +5,8 @@ import { CLI_BOOLEAN_FLAGS, CLI_GLOBAL_FLAGS, CLI_GLOBAL_VALUE_FLAGS, - findCliCommandIndex + findCliCommandIndex, + findCliCommandPathAt } from '../shared/cli-argument-boundary' export { specPaths } @@ -28,23 +29,60 @@ function setFlagValue( flags: Map, name: string, value: string, - search = false + repeatable: ReadonlySet ): void { const existing = flags.get(name) - if ( - typeof existing === 'string' && - (REPEATABLE_STRING_FLAGS.has(name) || (search && (name === 'agent' || name === 'path'))) - ) { + if (typeof existing === 'string' && repeatable.has(name)) { flags.set(name, `${existing}${REPEATED_FLAG_SEPARATOR}${value}`) return } flags.set(name, value) } -export function parseArgs(argv: string[], commandPaths?: readonly string[][]): ParsedArgs { +/** The most specific spec whose path prefixes `path`, so a group never shadows a leaf. */ +function specForPathPrefix( + specs: readonly CommandSpec[], + path: readonly string[] +): CommandSpec | undefined { + let best: { spec: CommandSpec; length: number } | undefined + for (const spec of specs) { + for (const candidate of specPaths(spec)) { + if ( + candidate.length <= path.length && + candidate.every((part, index) => part === path[index]) && + (!best || candidate.length > best.length) + ) { + best = { spec, length: candidate.length } + } + } + } + return best?.spec +} + +export function parseArgs( + argv: string[], + commandPaths?: readonly string[][], + specs: readonly CommandSpec[] = [] +): ParsedArgs { const commandPath: string[] = [] const flags = new Map() - const commandIndex = findCliCommandIndex(argv, commandPaths ?? []) + const paths = commandPaths ?? [] + // Why: the boundary scan and the flag reader must agree on which flags take no + // value, so both read the global set widened by every spec's own vocabulary. + const allBooleanFlags = new Set([ + ...BOOLEAN_FLAGS, + ...specs.flatMap((spec) => spec.booleanFlags ?? []) + ]) + const commandIndex = findCliCommandIndex(argv, paths, [], allBooleanFlags) + const pinned = + commandIndex === -1 ? null : findCliCommandPathAt(argv, paths, commandIndex, allBooleanFlags) + // Resolved lazily: without a registry the command is only known once its + // leading tokens have been read. + const activeSpec = (): CommandSpec | undefined => specForPathPrefix(specs, pinned ?? commandPath) + const repeatableFlags = (): ReadonlySet => { + const scoped = activeSpec()?.repeatableFlags + return scoped ? new Set([...REPEATABLE_STRING_FLAGS, ...scoped]) : REPEATABLE_STRING_FLAGS + } for (let i = 0; i < argv.length; i += 1) { const token = argv[i] @@ -63,19 +101,13 @@ export function parseArgs(argv: string[], commandPaths?: readonly string[][]): P flags, assignment.slice(0, equalsIndex), assignment.slice(equalsIndex + 1), - (argv[commandIndex] ?? commandPath[0]) === 'search' + repeatableFlags() ) continue } const flag = assignment - if ( - BOOLEAN_FLAGS.has(flag) || - ((argv[commandIndex] ?? commandPath[0]) === 'search' && - ['enable', 'disable', 'clear-index', 'index-status', 'pause', 'resume-indexing'].includes( - flag - )) - ) { + if (BOOLEAN_FLAGS.has(flag) || (activeSpec()?.booleanFlags?.includes(flag) ?? false)) { flags.set(flag, true) continue } @@ -90,7 +122,7 @@ export function parseArgs(argv: string[], commandPaths?: readonly string[][]): P flags.set(flag, true) continue } - setFlagValue(flags, flag, next, (argv[commandIndex] ?? commandPath[0]) === 'search') + setFlagValue(flags, flag, next, repeatableFlags()) i += 1 } diff --git a/src/cli/command-scoped-flag-parsing.test.ts b/src/cli/command-scoped-flag-parsing.test.ts new file mode 100644 index 00000000000..67642983ea9 --- /dev/null +++ b/src/cli/command-scoped-flag-parsing.test.ts @@ -0,0 +1,44 @@ +import { expect, it } from 'vitest' +import { parseArgs, REPEATED_FLAG_SEPARATOR, type CommandSpec } from './args' +import { COMMAND_SPECS } from './specs' + +// A command other than `search` on purpose: the parser must read this vocabulary +// off the resolved spec, not off a command name written into the parser. +const DEMO: CommandSpec = { + path: ['demo', 'run'], + summary: 'demo', + usage: 'demo run', + allowedFlags: ['enable', 'agent', 'note'], + booleanFlags: ['enable'], + repeatableFlags: ['agent'] +} + +it('reads value-less and repeatable flags from the spec that owns them', () => { + const parsed = parseArgs( + ['demo', 'run', '--enable', '--agent', 'codex', '--agent', 'claude', '--note', 'hi'], + [DEMO.path], + [DEMO] + ) + + expect(parsed.commandPath).toEqual(['demo', 'run']) + expect(parsed.flags.get('enable')).toBe(true) + expect(parsed.flags.get('agent')).toBe(`codex${REPEATED_FLAG_SEPARATOR}claude`) + expect(parsed.flags.get('note')).toBe('hi') +}) + +it('finds the command path behind a spec-declared boolean flag', () => { + const parsed = parseArgs(['--enable', 'demo', 'run'], [DEMO.path], [DEMO]) + + expect(parsed.commandPath).toEqual(['demo', 'run']) + expect(parsed.flags.get('enable')).toBe(true) +}) + +it('does not leak one command vocabulary into another', () => { + const parsed = parseArgs( + ['worktree', 'create', '--agent', 'codex', '--agent', 'claude'], + COMMAND_SPECS.flatMap((spec) => [spec.path]), + COMMAND_SPECS + ) + + expect(parsed.flags.get('agent')).toBe('claude') +}) diff --git a/src/cli/command-spec.ts b/src/cli/command-spec.ts index deba162d555..9a147d1725e 100644 --- a/src/cli/command-spec.ts +++ b/src/cli/command-spec.ts @@ -9,6 +9,11 @@ export type CommandSpec = { summary: string usage: string allowedFlags: string[] + // Why: value-less and repeatable flags are per-command vocabulary. Declaring + // them here keeps one command's flags out of the global parser, which cannot + // scope `--agent` (repeatable for `search`, single-valued for `worktree create`). + booleanFlags?: string[] + repeatableFlags?: string[] positionalArgs?: string[] examples?: string[] notes?: string[] diff --git a/src/cli/handlers/search.ts b/src/cli/handlers/search.ts index 1f71eea5991..dfaf151cabf 100644 --- a/src/cli/handlers/search.ts +++ b/src/cli/handlers/search.ts @@ -14,11 +14,11 @@ import { import { listSshTargets, findSshTargetByName } from '../host-selector-alternatives' import { searchAllHosts } from '../session-search-all-hosts' import { - searchHostMethod, + createSearchHostCall, SEARCH_ALL_TIMEOUT_MS, type SearchHost } from '../session-search-host-query' -import { waitForPromiseWithSignal } from '../../shared/abort-signal-reason' +import type { SessionSearchOperation } from '../../shared/ai-vault-search-rpc-methods' export const SEARCH_DISABLED_MESSAGE = 'Session search is off. Enable it in Settings > Agent Session History, or run `orca search --agent-session --enable`.' @@ -39,10 +39,6 @@ export const SEARCH_HANDLERS: Record = { controller.abort(new Error('Search interrupted.')) } process.once('SIGINT', interrupt) - const options = (): { signal: AbortSignal; timeoutMs: number } => ({ - signal: controller.signal, - timeoutMs: Math.max(1, deadline - Date.now()) - }) try { if (command.host === 'all') { const result = await searchAllHosts(client, command, controller.signal, deadline) @@ -96,16 +92,16 @@ export const SEARCH_HANDLERS: Record = { host.targetId = target.id host.name = target.label } - const target = host.targetId - ? { targetId: host.targetId } - : command.host?.kind === 'runtime' + // Why: `--host runtime:` already selected that runtime's transport, so + // the id only restamps the answer; the targetId spread lives in the shared + // call factory with the method routing and the deadline. + const stamp = + !host.targetId && command.host?.kind === 'runtime' ? { executionHostId: command.host.id } : {} - const call = (operation: 'query' | 'status' | 'configure', params: object) => - waitForPromiseWithSignal( - client.call(searchHostMethod(host, operation), { ...params, ...target }, options()), - controller.signal - ) + const send = createSearchHostCall(host, controller.signal, deadline) + const call = (operation: SessionSearchOperation, params: object) => + send(operation, { ...params, ...stamp }) if (command.configure || command.status) { const response = await call( command.configure ? 'configure' : 'status', diff --git a/src/cli/index.ts b/src/cli/index.ts index bcc1502226d..b1e4f6ece5b 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -83,7 +83,10 @@ export async function main( await runClaudeTeams(argv.slice(1), cwd) return } - const parsed = normalizeCommandPositionals(COMMAND_SPECS, parseArgs(argv, COMMAND_PATHS)) + const parsed = normalizeCommandPositionals( + COMMAND_SPECS, + parseArgs(argv, COMMAND_PATHS, COMMAND_SPECS) + ) const helpPath = resolveHelpPath(parsed) if (helpPath !== null) { printHelp(COMMAND_SPECS, helpPath) diff --git a/src/cli/runtime/transport.test.ts b/src/cli/runtime/transport.test.ts index 86015bf6a4a..257618294ec 100644 --- a/src/cli/runtime/transport.test.ts +++ b/src/cli/runtime/transport.test.ts @@ -92,36 +92,6 @@ describe.skipIf(process.platform === 'win32')('runtime transport', () => { } }) - it('rejects an oversized search frame before buffering the complete response', async () => { - const directory = mkdtempSync(join(tmpdir(), 'orca-search-size-')) - const endpoint = join(directory, 'runtime.sock') - const server = createServer((socket) => { - sockets.add(socket) - socket.on('error', () => undefined) - socket.once('close', () => sockets.delete(socket)) - socket.once('data', () => socket.write(Buffer.alloc(4 * 1024 * 1024 + 1, 'a'))) - }) - servers.add(server) - await new Promise((resolve) => server.listen(endpoint, resolve)) - try { - await expect( - sendRequest( - { - runtimeId: 'test', - pid: 1, - transports: [{ kind: 'unix', endpoint }], - authToken: 'fixture', - startedAt: 1 - }, - 'aiVault.searchSessions', - { query: 'fixture' }, - 30_000 - ) - ).rejects.toMatchObject({ code: 'invalid_runtime_response' }) - } finally { - rmSync(directory, { recursive: true, force: true }) - } - }) it('refreshes the per-call timeout when the runtime sends keepalive frames', async () => { const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-transport-')) const endpoint = join(userDataPath, 'runtime.sock') @@ -211,3 +181,46 @@ describe.skipIf(process.platform === 'win32')('runtime transport', () => { expect(Date.now() - start).toBeLessThan(5000) }) }) + +it('reads a large response frame on any method, without a per-method size rule', async () => { + const directory = mkdtempSync(join(tmpdir(), 'orca-search-size-')) + const endpoint = join(directory, 'runtime.sock') + const padding = 'a'.repeat(5 * 1024 * 1024) + const server = createServer((socket) => { + sockets.add(socket) + socket.on('error', () => undefined) + socket.once('close', () => sockets.delete(socket)) + let request = '' + socket.on('data', (chunk: Buffer) => { + request += chunk.toString('utf8') + const newline = request.indexOf('\n') + if (newline === -1) { + return + } + const { id } = JSON.parse(request.slice(0, newline)) + socket.write( + `${JSON.stringify({ id, ok: true, result: { padding }, _meta: { runtimeId: 'test' } })}\n` + ) + }) + }) + servers.add(server) + await new Promise((resolve) => server.listen(endpoint, resolve)) + try { + const response = await sendRequest<{ padding: string }>( + { + runtimeId: 'test', + pid: 1, + transports: [{ kind: 'unix', endpoint }], + authToken: 'fixture', + startedAt: 1 + }, + 'aiVault.searchSessions', + { query: 'fixture' }, + 30_000 + ) + expect(response.ok).toBe(true) + expect(response.ok === true && response.result.padding.length).toBe(padding.length) + } finally { + rmSync(directory, { recursive: true, force: true }) + } +}) diff --git a/src/cli/runtime/transport.ts b/src/cli/runtime/transport.ts index 92b659196ea..f3d13f7e250 100644 --- a/src/cli/runtime/transport.ts +++ b/src/cli/runtime/transport.ts @@ -35,9 +35,6 @@ export async function sendRequest( } const socket = createConnection(transport.endpoint) let lineSegments: string[] = [] - let lineBytes = 0 - const searchResponseLimit = - method.startsWith('aiVault.') && /search/i.test(method) ? 4 * 1024 * 1024 : Infinity let settled = false const requestId = randomUUID() @@ -80,10 +77,6 @@ export async function sendRequest( socket.destroy() } signal?.addEventListener('abort', onAbort, { once: true }) - if (signal?.aborted) { - onAbort() - return - } socket.setEncoding('utf8') socket.once('error', () => { finish({ @@ -116,20 +109,6 @@ export async function sendRequest( let cursor = 0 while (cursor < chunk.length && !settled) { const newlineIndex = chunk.indexOf('\n', cursor) - lineBytes += Buffer.byteLength( - chunk.slice(cursor, newlineIndex === -1 ? undefined : newlineIndex) - ) - if (lineBytes > searchResponseLimit) { - finish({ - ok: false, - error: new RuntimeClientError( - 'invalid_runtime_response', - 'Search response exceeds the size limit.' - ) - }) - socket.destroy() - return - } if (newlineIndex === -1) { lineSegments.push(chunk.slice(cursor)) return @@ -142,7 +121,6 @@ export async function sendRequest( lineSegments = [] } cursor = newlineIndex + 1 - lineBytes = 0 if (line.trim().length === 0) { continue } diff --git a/src/cli/search-command-arguments.test.ts b/src/cli/search-command-arguments.test.ts index 73042cdacf9..619e4f6c2d9 100644 --- a/src/cli/search-command-arguments.test.ts +++ b/src/cli/search-command-arguments.test.ts @@ -1,9 +1,15 @@ import { expect, it } from 'vitest' import { parseArgs, REPEATED_FLAG_SEPARATOR } from './args' import { parseSearchCommand } from './search-command-arguments' +import { SEARCH_COMMAND_SPECS } from './specs/search' + +// The search flag vocabulary lives on its spec, so the parser only knows the +// repeatable and value-less flags when the registry is handed to it. +const parseSearchArgs = (argv: string[]): ReturnType => + parseArgs(argv, [['search']], SEARCH_COMMAND_SPECS) it('preserves repeated filters through argv and validates before configuration', () => { - const parsed = parseArgs([ + const parsed = parseSearchArgs([ 'search', '--agent-session', 'needle', @@ -34,33 +40,32 @@ it('preserves repeated filters through argv and validates before configuration', it('accepts queryless policy management and refuses aggregate mutations', () => { expect( - parseSearchCommand(parseArgs(['search', '--agent-session', '--enable']).flags).configure + parseSearchCommand(parseSearchArgs(['search', '--agent-session', '--enable']).flags).configure ).toEqual({ enabled: true }) expect( parseSearchCommand( - parseArgs(['search', '--disable', '--clear-index', '--host', 'ssh:box']).flags + parseSearchArgs(['search', '--disable', '--clear-index', '--host', 'ssh:box']).flags ).configure ).toEqual({ enabled: false, clearIndex: true }) expect(() => - parseSearchCommand(parseArgs(['search', '--enable', '--host', 'all']).flags) + parseSearchCommand(parseSearchArgs(['search', '--enable', '--host', 'all']).flags) + ).toThrow() + expect(() => + parseSearchCommand(parseSearchArgs(['search', '--enable', '--disable']).flags) ).toThrow() - expect(() => parseSearchCommand(parseArgs(['search', '--enable', '--disable']).flags)).toThrow() }) it('handles command discovery, equals syntax, Windows paths and host-specific scope rules', () => { - const parsed = parseArgs( - [ - '--json', - 'search', - '--agent-session=needle', - '--agent=codex', - '--agent=claude', - '--path=C:\\work', - '--path=\\\\server\\share', - '--host=all' - ], - [['search']] - ) + const parsed = parseSearchArgs([ + '--json', + 'search', + '--agent-session=needle', + '--agent=codex', + '--agent=claude', + '--path=C:\\work', + '--path=\\\\server\\share', + '--host=all' + ]) expect(parseSearchCommand(parsed.flags).query).toMatchObject({ agents: ['codex', 'claude'], scopePaths: ['C:\\work', '\\\\server\\share'] @@ -72,6 +77,6 @@ it('handles command discovery, equals syntax, Windows paths and host-specific sc ['--agent-session=needle', '--host=all', '--path=~/private'], ['--index-status', '--agent-session=needle'] ]) { - expect(() => parseSearchCommand(parseArgs(['search', ...args], [['search']]).flags)).toThrow() + expect(() => parseSearchCommand(parseSearchArgs(['search', ...args]).flags)).toThrow() } }) diff --git a/src/cli/session-search-all-hosts.ts b/src/cli/session-search-all-hosts.ts index 1287178ac34..a7e81ca0b19 100644 --- a/src/cli/session-search-all-hosts.ts +++ b/src/cli/session-search-all-hosts.ts @@ -137,29 +137,25 @@ export async function searchAllHosts( message: 'Overall search deadline exceeded.' } } - const controller = new AbortController() - const timer = setTimeout( - () => controller.abort(new Error('Search host deadline exceeded.')), - Math.min(SEARCH_HOST_TIMEOUT_MS, deadline - Date.now()) + // Why: querySearchHost owns the per-host deadline; arming a second one here + // produced two timers and two error messages for one call. + const result = await querySearchHost( + host, + command, + true, + signal, + Math.min(deadline, Date.now() + SEARCH_HOST_TIMEOUT_MS) ) - const abort = (): void => controller.abort(signal.reason) - signal.addEventListener('abort', abort, { once: true }) - try { - const result = await querySearchHost(host, command, true, controller.signal) - const bytes = Buffer.byteLength(JSON.stringify(result)) - if (bytes > bytesRemaining) { - return { - host: result.host, - outcome: 'omitted', - message: 'Aggregate response limit reached.' - } + const bytes = Buffer.byteLength(JSON.stringify(result)) + if (bytes > bytesRemaining) { + return { + host: result.host, + outcome: 'omitted', + message: 'Aggregate response limit reached.' } - bytesRemaining -= bytes - return result - } finally { - clearTimeout(timer) - signal.removeEventListener('abort', abort) } + bytesRemaining -= bytes + return result }) const combined = [...results, ...skipped] if ( diff --git a/src/cli/session-search-host-query.ts b/src/cli/session-search-host-query.ts index db95aa7d519..fedd0e51ec5 100644 --- a/src/cli/session-search-host-query.ts +++ b/src/cli/session-search-host-query.ts @@ -1,9 +1,14 @@ import type { RuntimeClient } from './runtime-client' +import type { RuntimeRpcSuccess } from './runtime/types' import { SessionSearchResultSchema, SessionSearchStatusSchema } from '../shared/ai-vault-search-contract' import type { AiVaultSearchResult } from '../shared/ai-vault-search-types' +import { + SESSION_SEARCH_METHODS, + type SessionSearchOperation +} from '../shared/ai-vault-search-rpc-methods' import type { SearchCommand } from './search-command-arguments' import { waitForPromiseWithSignal } from '../shared/abort-signal-reason' @@ -33,46 +38,53 @@ export const SEARCH_ALL_HOST_LIMIT = 16 export function searchHostMethod( host: Pick, - operation: 'query' | 'status' | 'configure' + operation: SessionSearchOperation ): string { - if (host.targetId) { - return `aiVault.sshSearch${{ query: 'Sessions', status: 'IndexStatus', configure: 'Configure' }[operation]}` + const methods = SESSION_SEARCH_METHODS[operation] + return host.targetId ? methods.runtimeSsh : methods.runtime +} + +/** + * The one place that knows a search host's method routing, its target spread and + * its deadline, so the single-host and all-hosts paths cannot arm two of any of + * them for the same call. + */ +export function createSearchHostCall( + host: Pick, + signal: AbortSignal, + deadline: number +): (operation: SessionSearchOperation, params?: object) => Promise> { + return (operation, params = {}) => { + const remaining = deadline - Date.now() + if (remaining <= 0) { + return Promise.reject(new Error('Search host deadline exceeded.')) + } + return waitForPromiseWithSignal( + host.client.call( + searchHostMethod(host, operation), + { ...params, ...(host.targetId ? { targetId: host.targetId } : {}) }, + { timeoutMs: remaining, signal } + ), + signal + ) } - return { - query: 'aiVault.searchSessions', - status: 'aiVault.searchIndexStatus', - configure: 'aiVault.configureSessionSearch' - }[operation] } export async function querySearchHost( host: SearchHost, command: SearchCommand, aggregate: boolean, - signal: AbortSignal + signal: AbortSignal, + deadline = Date.now() + SEARCH_HOST_TIMEOUT_MS ): Promise { const identity: SearchHostResult['host'] = { id: host.id, name: host.name, selector: host.selector } - const deadline = Date.now() + SEARCH_HOST_TIMEOUT_MS - const call = async (operation: 'query' | 'status', args: object = {}): Promise => { - const remaining = deadline - Date.now() - if (remaining <= 0) { - throw new Error('Search host deadline exceeded.') - } - const response = await waitForPromiseWithSignal( - host.client.call( - searchHostMethod(host, operation), - { - ...args, - ...(host.targetId ? { targetId: host.targetId } : {}) - }, - { timeoutMs: remaining, signal } - ), - signal - ) + const send = createSearchHostCall(host, signal, deadline) + const call = async (operation: SessionSearchOperation, args: object = {}): Promise => { + const response = await send(operation, args) if (response._meta?.runtimeId) { identity.runtimeId = response._meta.runtimeId.slice(0, 512) } diff --git a/src/cli/specs/search.ts b/src/cli/specs/search.ts index d42d3d3be39..23c421e6770 100644 --- a/src/cli/specs/search.ts +++ b/src/cli/specs/search.ts @@ -23,6 +23,16 @@ export const SEARCH_COMMAND_SPECS: CommandSpec[] = [ 'pause', 'resume-indexing' ], + booleanFlags: [ + 'enable', + 'disable', + 'clear-index', + 'index-status', + 'pause', + 'resume-indexing', + 'newest' + ], + repeatableFlags: ['agent', 'path'], notes: [ 'Searches what you typed, what the agent said, the commands it ran, and the first 3 KB of each tool output across Claude Code, Codex, Cursor, Gemini, OpenCode, and the other agents Orca scans.', 'Quote paths, identifiers, or error text to match them exactly; plain words match anywhere. A misspelled word is repaired from the index vocabulary when nothing matches.', diff --git a/src/main/ai-vault-search/session-search-cancel-publication.test.ts b/src/main/ai-vault-search/session-search-cancel-publication.test.ts index 58cbb4a354b..daeb289441e 100644 --- a/src/main/ai-vault-search/session-search-cancel-publication.test.ts +++ b/src/main/ai-vault-search/session-search-cancel-publication.test.ts @@ -5,7 +5,7 @@ import { tmpdir } from 'node:os' import { SessionSearchStore } from './session-search-store' import { parseSearchCandidates } from './session-search-parse-candidates' import { sessionCandidate } from './session-search-transcript-fixtures' -import { stagedWriteUpdate } from './session-search-staged-write-fixtures' +import { stagedWriteUpdate } from './session-search-staged-write-test-fixture' import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture' import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' import * as sourceRead from '../native-chat/wsl-transcript-fs-access' @@ -38,7 +38,9 @@ it('preserves published content and cursor when a whole-JSON refresh is canceled return text }) const controller = new AbortController() - parsing = parseSearchCandidates(store, [candidate], controller.signal).catch((error) => error) + parsing = parseSearchCandidates(store, [candidate], { signal: controller.signal }).catch( + (error) => error + ) await reached.promise controller.abort() held.resolve() diff --git a/src/main/ai-vault-search/session-search-capture-regressions.test.ts b/src/main/ai-vault-search/session-search-capture-regressions.test.ts index fbedc55fe74..05e5f929e63 100644 --- a/src/main/ai-vault-search/session-search-capture-regressions.test.ts +++ b/src/main/ai-vault-search/session-search-capture-regressions.test.ts @@ -2,6 +2,7 @@ import { appendFile, mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' import { join } from 'node:path' import { tmpdir } from 'node:os' import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import SyncDatabase from '../sqlite/sync-database' import { SessionSearchStore } from './session-search-store' import { SessionSearchService } from './session-search-service' import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture' @@ -21,10 +22,12 @@ import { } from './session-search-transcript-fixtures' let root: string let store: SessionSearchStore +let databasePath: string beforeEach(async () => { resetSessionParseCacheForTests() root = await mkdtemp(join(tmpdir(), 'ss-capture-audit-')) - store = new SessionSearchStore(join(root, 'index.sqlite')) + databasePath = join(root, 'index.sqlite') + store = new SessionSearchStore(databasePath) registerSessionSearchIndexSink(store) }) afterEach(async () => { @@ -61,15 +64,36 @@ it('refreshes indexed Codex metadata when the title index changes', async () => expect(listed?.title).toBe('Renamed synthetic title') expect(store.search({ query: 'needle' }).hits[0]?.title).toBe('Renamed synthetic title') }) +it('does not rewrite indexed Codex metadata when the refresh finds no change', async () => { + const path = join(root, CODEX_ROLLOUT_FILE) + await writeFile( + path, + `${codexRolloutLines(['echo'], 'synthetic output', 'synthetic needle').join('\n')}\n` + ) + const candidate = await sessionCandidate('codex', path, root) + await parseAgentSessionFileCached(candidate, process.platform) + // A cache hit still re-reads the Codex title index; an unchanged read must + // not issue an UPDATE, because list scans repeat every few seconds. + const update = vi.spyOn(store, 'updateMetadata') + await parseAgentSessionFileCached(candidate, process.platform) + expect(update).not.toHaveBeenCalled() +}) it('redacts credential-shaped content copied into session titles', async () => { const fakeKey = `sk-${'x'.repeat(40)}` const path = join(root, `${CLAUDE_SESSION_ID}.jsonl`) await writeFile(path, `${userRecord(0, `synthetic needle ${fakeKey}`)}\n`) await parseAgentSessionFileCached(await sessionCandidate('claude', path), process.platform) - const body = store.db.prepare('SELECT user_text FROM messages_fts').get() as { user_text: string } - expect(body.user_text).not.toContain(fakeKey) - const row = store.db.prepare('SELECT title FROM sessions').get() as { title: string } - expect(row.title).not.toContain(fakeKey) + const reader = new SyncDatabase(databasePath, { readonly: true }) + try { + const body = reader.prepare('SELECT user_text FROM messages_fts').get() as { + user_text: string + } + expect(body.user_text).not.toContain(fakeKey) + const row = reader.prepare('SELECT title FROM sessions').get() as { title: string } + expect(row.title).not.toContain(fakeKey) + } finally { + reader.close() + } }) it('does not log malformed JSON transcript excerpts during backfill', async () => { const roots = isolatedScanRoots(root) @@ -92,6 +116,6 @@ it('does not log malformed JSON transcript excerpts during backfill', async () = .join(' ') expect(logged).not.toContain('private_sy') } finally { - service.dispose() + await service.close() } }) diff --git a/src/main/ai-vault-search/session-search-consent.test.ts b/src/main/ai-vault-search/session-search-consent.test.ts index 07069b72775..5adad8ae284 100644 --- a/src/main/ai-vault-search/session-search-consent.test.ts +++ b/src/main/ai-vault-search/session-search-consent.test.ts @@ -54,6 +54,7 @@ import type * as ParseCachePersistence from '../ai-vault/session-parse-cache-per import type * as SourceDiscovery from '../ai-vault/session-scanner-source-discovery' import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' import { isolatedScanRoots, jsonLines } from '../ai-vault/session-scanner-test-fixtures' +import * as sourcePresence from './session-search-source-presence' import { SessionSearchService, type SessionSearchScanRoots } from './session-search-service' let tempRoots: string[] = [] @@ -66,9 +67,8 @@ beforeEach(() => { }) afterEach(async () => { - for (const service of services) { - service.dispose() - } + vi.restoreAllMocks() + await Promise.all(services.map((service) => service.close())) services = [] await Promise.all(tempRoots.map((root) => rm(root, { recursive: true, force: true }))) tempRoots = [] @@ -345,6 +345,74 @@ describe('SessionSearchService consent gate', () => { ).toEqual(['recent-session']) }) + it('answers with no hits when a disable closes the index between query rounds', async () => { + const { roots, databasePath } = await scanRoots() + await writeClaudeTranscript(roots, 'closed-session', 'the vacuum quota never settles') + const service = makeService(databasePath, { enabled: true, historyDays: null }) + await service.ensureBackfill(roots) + + // The presence pass awaits a stat between rounds; a disable landing there + // used to leave the resumed round querying a closed database handle. + let indexedHits = 0 + vi.spyOn(sourcePresence, 'searchPresentSessionSources').mockImplementation( + async (args, search) => { + indexedHits = search(args).hits.length + await service.configure({ enabled: false, historyDays: null }, roots) + return search(args) + } + ) + + const result = await service.search({ query: 'vacuum', refresh: false }, roots) + + expect(indexedHits).toBeGreaterThan(0) + expect(result.hits).toEqual([]) + }) + + it('waits for a parked backfill before the shutdown drops the sink', async () => { + const { roots, databasePath } = await scanRoots() + await writeClaudeTranscript(roots, 'draining-session', 'the vacuum quota never settles') + const service = makeService(databasePath, { enabled: true, historyDays: null }) + let releaseBackfill!: () => void + holdNextParseCacheLoad = new Promise((resolve) => { + releaseBackfill = resolve + }) + const backfill = service.ensureBackfill(roots) + await vi.waitFor(() => expect(parseCacheLoads).toBe(1)) + + // A parse parked in the backfill must finish before the store closes under + // it, so the sink survives until the drain completes. + let closed = false + const closing = service.close().then(() => { + closed = true + }) + await new Promise((resolve) => setImmediate(resolve)) + expect(closed).toBe(false) + expect(getSessionSearchIndexSink()).not.toBeNull() + + releaseBackfill() + await closing + await backfill + expect(getSessionSearchIndexSink()).toBeNull() + }) + + it("leaves a replacement service's sink registered when the old one closes late", async () => { + const { roots, databasePath } = await scanRoots() + await writeClaudeTranscript(roots, 'handover-session', 'the vacuum quota never settles') + const retiring = makeService(databasePath, { enabled: true, historyDays: null }) + await retiring.ensureBackfill(roots) + + // Shutdown is async, so a replacement can claim the process-global sink + // first; the late close must not clear a sink it no longer owns. + const replacement = makeService(databasePath, { enabled: true, historyDays: null }) + await retiring.close() + + expect(getSessionSearchIndexSink()).not.toBeNull() + await replacement.ensureBackfill(roots) + expect( + (await replacement.search({ query: 'vacuum', refresh: false }, roots)).hits.length + ).toBeGreaterThan(0) + }) + it('skips transcripts older than the history bound', async () => { const { roots, databasePath } = await scanRoots() await writeClaudeTranscript(roots, 'recent-session', 'the vacuum quota never settles', 1) diff --git a/src/main/ai-vault-search/session-search-durable-reuse.test.ts b/src/main/ai-vault-search/session-search-durable-reuse.test.ts index ed4537d2c3c..a55aef39cb8 100644 --- a/src/main/ai-vault-search/session-search-durable-reuse.test.ts +++ b/src/main/ai-vault-search/session-search-durable-reuse.test.ts @@ -2,7 +2,9 @@ import { mkdtemp, rm, stat, writeFile } from 'node:fs/promises' import { join } from 'node:path' import { tmpdir } from 'node:os' import { afterEach, expect, it, vi } from 'vitest' +import SyncDatabase from '../sqlite/sync-database' import { SessionSearchStore } from './session-search-store' +import { openSessionSearchDatabase } from './session-search-schema' import { parseSearchCandidates } from './session-search-parse-candidates' import { userRecord, @@ -23,10 +25,20 @@ import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' vi.mock('./session-search-backfill-pacing', () => ({ pauseBackfill: async () => {} })) let directory: string | undefined +let databasePath = '' let store: SessionSearchStore | undefined +let reader: SyncDatabase | undefined + +/** The store keeps its connection private; row assertions read the same file separately. */ +function rows(sql: string, ...values: unknown[]): unknown[] { + reader ??= new SyncDatabase(databasePath, { readonly: true }) + return reader.prepare(sql).all(...(values as never[])) +} afterEach(async () => { registerSessionSearchIndexSink(null) + reader?.close() + reader = undefined store?.close() resetSessionParseCacheForTests() resetCodexSessionIndexTitleCacheForTests() @@ -43,7 +55,7 @@ async function fixture() { `${userRecord(0, 'ordinary title')}\n${assistantRecord(1, 'durableneedle')}\n` ) const candidate = await candidateAt(path) - const databasePath = join(directory, 'index.sqlite') + databasePath = join(directory, 'index.sqlite') store = new SessionSearchStore(databasePath) registerSessionSearchIndexSink(store) await parseSearchCandidates(store, [candidate]) @@ -73,14 +85,11 @@ async function candidateAt(path: string): Promise { it('reuses an unchanged reopened index without a preview cache or replacement write', async () => { const { candidate, store } = await fixture() const apply = vi.spyOn(store, 'apply') - const before = store.db - .prepare('SELECT session_row_id FROM files WHERE path=?') - .get(candidate.file.path) + const fileRow = 'SELECT session_row_id FROM files WHERE path=?' + const before = rows(fileRow, candidate.file.path) await parseSearchCandidates(store, [candidate]) expect(apply).not.toHaveBeenCalled() - expect( - store.db.prepare('SELECT session_row_id FROM files WHERE path=?').get(candidate.file.path) - ).toEqual(before) + expect(rows(fileRow, candidate.file.path)).toEqual(before) expect(store.search({ query: 'durableneedle' }).hits).toHaveLength(1) // Listing still needs its preview, even when backfill can reuse the index. expect( @@ -102,7 +111,10 @@ it('does not skip a changed or atomically replaced transcript', async () => { }) it('keeps durable reuse beyond the 4096-entry preview cache', async () => { - store = new SessionSearchStore(':memory:') + directory = await mkdtemp(join(tmpdir(), 'search-durable-cache-')) + databasePath = join(directory, 'index.sqlite') + const seedConnection = openSessionSearchDatabase(databasePath) + store = new SessionSearchStore(databasePath) registerSessionSearchIndexSink(store) const now = Date.now() const candidates: SessionFileCandidate[] = Array.from({ length: 4097 }, (_, i) => ({ @@ -117,14 +129,15 @@ it('keeps durable reuse beyond the 4096-entry preview cache', async () => { ino: i + 1 } })) - store.db.exec('BEGIN') - const insert = store.db.prepare( + seedConnection.exec('BEGIN') + const insert = seedConnection.prepare( 'INSERT INTO files(path,dev,ino,mtime_ms,size_bytes,byte_offset) VALUES(?,?,?,?,?,?)' ) for (const { file } of candidates) { insert.run(file.path, file.dev!, file.ino!, now, 1, 1) } - store.db.exec('COMMIT') + seedConnection.exec('COMMIT') + seedConnection.close() seedSessionParseCache( candidates.map(({ file }) => [ file.path, @@ -142,11 +155,14 @@ it('refreshes external Codex titles on cold reuse without replacing transcript r const path = join(directory, CODEX_ROLLOUT_FILE) await writeFile(path, `${codexRolloutLines(['echo'], 'output', 'titleneedle').join('\n')}\n`) const candidate = await sessionCandidate('codex', path, directory) - const databasePath = join(directory, 'index.sqlite') + databasePath = join(directory, 'index.sqlite') store = new SessionSearchStore(databasePath) registerSessionSearchIndexSink(store) await parseSearchCandidates(store, [candidate]) - const before = store.db.prepare('SELECT id FROM sessions WHERE index_ready = 1').all() + const readyRows = 'SELECT id FROM sessions WHERE index_ready = 1' + const before = rows(readyRows) + reader?.close() + reader = undefined store.close() resetSessionParseCacheForTests() resetCodexSessionIndexTitleCacheForTests() @@ -163,5 +179,5 @@ it('refreshes external Codex titles on cold reuse without replacing transcript r await parseSearchCandidates(store, [candidate]) expect(store.search({ query: 'titleneedle' }).hits[0].title).toBe('Renamed durable title') expect(apply).not.toHaveBeenCalled() - expect(store.db.prepare('SELECT id FROM sessions WHERE index_ready = 1').all()).toEqual(before) + expect(rows(readyRows)).toEqual(before) }) diff --git a/src/main/ai-vault-search/session-search-enablement.test.ts b/src/main/ai-vault-search/session-search-enablement.test.ts index d090990a7f4..530e8618a42 100644 --- a/src/main/ai-vault-search/session-search-enablement.test.ts +++ b/src/main/ai-vault-search/session-search-enablement.test.ts @@ -4,6 +4,7 @@ import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { applyAiVaultSearchSettings, + applyAiVaultSearchSettingsChange, clearAiVaultSearchIndex, installAiVaultSearchSettingsSource, readAiVaultSearchIndexStatus @@ -213,3 +214,68 @@ it('requires desktop clear to retry a failed policy flush before reporting appli await clearAiVaultSearchIndex(persist) expect(readAiVaultSearchIndexStatus().applied).toBe(true) }) + +describe('applyAiVaultSearchSettingsChange', () => { + it('does not reconfigure the scanner when the saved policy is unchanged', async () => { + initSessionSearchPaths(await makeUserDataDir()) + const settings = { aiVaultSearch: { enabled: true, historyDays: 90 } } + const persist = vi.fn() + + applyAiVaultSearchSettingsChange( + settings, + { aiVaultSearch: { ...settings.aiVaultSearch } }, + persist + ) + // The apply chain is shared and serialized, so awaiting a later apply proves + // the unchanged write never queued one of its own. + await applyAiVaultSearchSettings({ aiVaultSearch: { enabled: false, historyDays: 90 } }) + + expect(configureAiVaultSearch).toHaveBeenCalledTimes(1) + expect(configureAiVaultSearch).toHaveBeenCalledWith( + expect.objectContaining({ enabled: false }), + expect.anything() + ) + expect(persist).not.toHaveBeenCalled() + }) + + it('forwards a pause that leaves consent and retention alone', async () => { + initSessionSearchPaths(await makeUserDataDir()) + applyAiVaultSearchSettingsChange( + { aiVaultSearch: { enabled: true, historyDays: 90 } }, + { aiVaultSearch: { enabled: true, historyDays: 90, paused: true } }, + () => undefined + ) + + await vi.waitFor(() => + expect(configureAiVaultSearch).toHaveBeenCalledWith( + expect.objectContaining({ paused: true }), + expect.anything() + ) + ) + }) + + it('reports a failed apply through the index status instead of throwing at the caller', async () => { + initSessionSearchPaths(await makeUserDataDir()) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + configureAiVaultSearch.mockRejectedValueOnce(new Error('scanner unavailable')) + try { + applyAiVaultSearchSettingsChange( + { aiVaultSearch: { enabled: false, historyDays: null } }, + { aiVaultSearch: { enabled: true, historyDays: null } }, + () => undefined + ) + await vi.waitFor(() => expect(readAiVaultSearchIndexStatus().applied).toBe(false)) + expect(readAiVaultSearchIndexStatus().reason).toContain('failed or is pending') + // The scanner failure must have been absorbed, not left for the caller. + await vi.waitFor(() => + expect(warn).toHaveBeenCalledWith( + '[settings] failed to apply agent session search settings:', + expect.any(Error) + ) + ) + } finally { + warn.mockRestore() + } + await applyAiVaultSearchSettings({ aiVaultSearch: { enabled: true, historyDays: null } }) + }) +}) diff --git a/src/main/ai-vault-search/session-search-enablement.ts b/src/main/ai-vault-search/session-search-enablement.ts index 4193d3758b3..12446853524 100644 --- a/src/main/ai-vault-search/session-search-enablement.ts +++ b/src/main/ai-vault-search/session-search-enablement.ts @@ -62,6 +62,31 @@ let applyChain: Promise = Promise.resolve() let policyApplied = true let applyGeneration = 0 +/** + * Reconciles a settings write. An unchanged policy is not forwarded, so re-saving + * the same value never restarts a running backfill, and a scanner that cannot + * apply must not fail or delay the settings save — `readAiVaultSearchIndexStatus` + * reports that through `applied` and `reason`. + */ +export function applyAiVaultSearchSettingsChange( + before: Pick, + after: Pick, + persist: () => void | Promise +): void { + const previous = resolveAiVaultSearchSettings(before) + const next = resolveAiVaultSearchSettings(after) + if ( + previous.enabled === next.enabled && + previous.historyDays === next.historyDays && + (previous.paused ?? false) === (next.paused ?? false) + ) { + return + } + void applyAiVaultSearchSettings(after, { persist }).catch((error: unknown) => { + console.warn('[settings] failed to apply agent session search settings:', error) + }) +} + export function readAiVaultSearchIndexStatus(): AiVaultSearchIndexStatus { const capability = sessionSearchCapability() const available = diff --git a/src/main/ai-vault-search/session-search-fts5-contract.test.ts b/src/main/ai-vault-search/session-search-fts5-contract.test.ts index a95c0fca892..36bad9280ba 100644 --- a/src/main/ai-vault-search/session-search-fts5-contract.test.ts +++ b/src/main/ai-vault-search/session-search-fts5-contract.test.ts @@ -6,6 +6,9 @@ import { removeTree } from '../../shared/windows-transient-lock-removal' import type SyncDatabase from '../sqlite/sync-database' import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture' +import { indexTokens } from './session-search-query-planner' +import { sessionRowFilter } from './session-search-row-filter' +import { splitAiVaultSearchQuery } from '../../shared/ai-vault-search-query-operators' import { openSessionSearchDatabase } from './session-search-schema' import { SessionSearchStore } from './session-search-store' import { parseTranscript as parse, userRecord } from './session-search-transcript-fixtures' @@ -233,3 +236,43 @@ describe('SessionSearchStore.search snippets', () => { store.close() }) }) + +describe('the planner tokenizer draws the same boundaries as unicode61', () => { + // unicode61 folds case and strips Latin diacritics on both index and query side. + function asIndexed(token: string): string { + return token.toLowerCase().normalize('NFD').replaceAll(/\p{M}/gu, '') + } + + it('produces exactly the terms fts5vocab reports for the same text', async () => { + const db = await openDatabase() + const corpus = + 'resolveTerminalPath src/main/foo-bar.ts a.b C++ #123 修复 café naïve MAX_TOKEN x' + insertMessageRow(db, FIRST_ROWID, corpus) + const indexed = ( + db.prepare('SELECT term FROM messages_vocab ORDER BY term').all() as { term: string }[] + ).map((row) => row.term) + + expect([...new Set(indexTokens(corpus).map(asIndexed))].sort()).toEqual(indexed) + }) +}) + +describe('a cwd scope seeks the cwd_key index instead of scanning it', () => { + it('plans the scope condition as a SEARCH on sessions_cwd_key', async () => { + const db = await openDatabase() + const filter = sessionRowFilter( + { query: 'needle', scopePaths: ['/work/app'] }, + splitAiVaultSearchQuery('needle') + ) + const plan = ( + db + .prepare( + `EXPLAIN QUERY PLAN SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')}` + ) + .all(...filter.values) as { detail: string }[] + ).map((row) => row.detail) + + expect(plan.join(' | ')).toContain('sessions_cwd_key') + expect(plan.some((detail) => detail.startsWith('SEARCH'))).toBe(true) + expect(plan.some((detail) => detail.startsWith('SCAN sessions'))).toBe(false) + }) +}) diff --git a/src/main/ai-vault-search/session-search-hit-ranking.ts b/src/main/ai-vault-search/session-search-hit-ranking.ts new file mode 100644 index 00000000000..6a61c4fd425 --- /dev/null +++ b/src/main/ai-vault-search/session-search-hit-ranking.ts @@ -0,0 +1,139 @@ +import type { AiVaultAgent } from '../../shared/ai-vault-types' +import type { AiVaultSearchArgs, AiVaultSearchHit } from '../../shared/ai-vault-search-types' +import { + AI_VAULT_SEARCH_LIMIT_DEFAULT, + AI_VAULT_SEARCH_LIMIT_MAX +} from '../../shared/ai-vault-search-types' +import { isCollapsibleContentHash } from './session-search-content-hash' + +// Subtracted per session: `0.02 · ln(1 + messages)`; slightly positive on both eval sets. +const LENGTH_PRIOR = 0.02 + +export type SessionRow = { + id: number + agent: AiVaultAgent + session_id: string + file_path: string + codex_home: string | null + title: string + cwd: string | null + branch: string | null + updated_at: string | null + message_count: number + resume_command: string + content_hash: string | null + content_hash_count: number +} + +/** The one message that stands for a session: its best-scoring match. */ +export type MessageRow = { + rowid: number + score: number + session_row_id: number + role: string + ts: string | null +} + +type ScoredSession = { + session: SessionRow + message: MessageRow + score: number + duplicateCount: number +} + +export function sessionFields(session: SessionRow): Omit { + return { + agent: session.agent, + sessionId: session.session_id, + filePath: session.file_path, + codexHome: session.codex_home, + title: session.title, + cwd: session.cwd, + branch: session.branch, + updatedAt: session.updated_at, + messageCount: session.message_count, + resumeCommand: session.resume_command + } +} + +// Why: the desktop IPC forwards its payload unvalidated, so a non-positive +// limit must be clamped here or `LIMIT -1` / `slice(0, -1)` leak through. +export function resolveLimit(args: AiVaultSearchArgs): number { + const requested = Number.isInteger(args.limit) + ? (args.limit as number) + : AI_VAULT_SEARCH_LIMIT_DEFAULT + return Math.min(Math.max(1, requested), AI_VAULT_SEARCH_LIMIT_MAX) +} + +/** + * Everything between "these sessions matched" and "this is the page": the length + * prior, fork folding, the caller's order, and the cut. Retrieval stays in SQL; + * nothing here touches the database, and `snippet` runs only for the page. + */ +export function rankSessionHits( + sessions: readonly SessionRow[], + matches: ReadonlyMap, + args: AiVaultSearchArgs, + snippet: (message: MessageRow) => string +): AiVaultSearchHit[] { + const scored = collapseForks( + sessions.map((session) => { + const message = matches.get(session.id)! + return { + session, + message, + score: message.score - LENGTH_PRIOR * Math.log(1 + session.message_count), + duplicateCount: 1 + } + }) + ) + scored.sort((left, right) => + args.sort === 'newest' + ? (right.session.updated_at ?? '').localeCompare(left.session.updated_at ?? '') + : right.score - left.score + ) + return scored.slice(0, resolveLimit(args)).map(({ session, message, score, duplicateCount }) => ({ + ...sessionFields(session), + score, + ...(duplicateCount > 1 ? { duplicateCount } : {}), + evidence: { + role: message.role as AiVaultSearchHit['evidence']['role'], + timestamp: message.ts, + snippet: snippet(message) + } + })) +} + +/** + * Folds forked copies of one conversation into a single hit: same opening + * prefix, newest `updated_at` wins, the rest become `duplicateCount`. Done here + * and not at write time so index rows stay per file (cursors and deletes). + */ +function collapseForks(scored: ScoredSession[]): ScoredSession[] { + const groups = new Map() + for (const entry of scored) { + const { content_hash: hash, content_hash_count: count, id } = entry.session + const key = isCollapsibleContentHash(hash, count) ? `hash:${hash}` : `session:${id}` + const group = groups.get(key) + if (group) { + group.push(entry) + } else { + groups.set(key, [entry]) + } + } + const collapsed: ScoredSession[] = [] + for (const group of groups.values()) { + if (group.length === 1) { + collapsed.push(group[0]!) + continue + } + const winner = group.reduce((best, entry) => (isNewer(entry, best) ? entry : best)) + collapsed.push({ ...winner, duplicateCount: group.length }) + } + return collapsed +} + +function isNewer(entry: ScoredSession, best: ScoredSession): boolean { + const order = (entry.session.updated_at ?? '').localeCompare(best.session.updated_at ?? '') + return order === 0 ? entry.score > best.score : order > 0 +} diff --git a/src/main/ai-vault-search/session-search-index-compaction.ts b/src/main/ai-vault-search/session-search-index-compaction.ts new file mode 100644 index 00000000000..93d2b2649d0 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-compaction.ts @@ -0,0 +1,22 @@ +import { setImmediate as yieldToEventLoop } from 'node:timers/promises' +import type SyncDatabase from '../sqlite/sync-database' + +const COMPACT_PAGES_PER_STEP = 2000 + +/** Hands freed pages back to the filesystem in bounded steps, never one long stall. */ +export async function compactSessionSearchIndex( + db: SyncDatabase, + stopped: () => boolean +): Promise { + let freed = Number(db.pragma('freelist_count', { simple: true })) + while (!stopped() && freed > 0) { + db.pragma(`incremental_vacuum(${COMPACT_PAGES_PER_STEP})`) + const remaining = Number(db.pragma('freelist_count', { simple: true })) + // Why: without auto_vacuum the step is a no-op; never spin on it. + if (remaining >= freed) { + return + } + freed = remaining + await yieldToEventLoop() + } +} diff --git a/src/main/ai-vault-search/session-search-index-writer.ts b/src/main/ai-vault-search/session-search-index-writer.ts index 03c6c978bd6..d267d89d884 100644 --- a/src/main/ai-vault-search/session-search-index-writer.ts +++ b/src/main/ai-vault-search/session-search-index-writer.ts @@ -1,4 +1,4 @@ -import { assertSearchWalBudget } from './session-search-wal-budget' +import { assertSearchWalBudget, SEARCH_WAL_PENDING_BYTES } from './session-search-wal-budget' import { setImmediate as yieldToEventLoop } from 'node:timers/promises' import type { AiVaultSession } from '../../shared/ai-vault-types' import type SyncDatabase from '../sqlite/sync-database' @@ -11,13 +11,26 @@ import type { import { EMPTY_CONTENT_HASH, foldContentHash } from './session-search-content-hash' import { SessionSearchFileRecords } from './session-search-file-records' import { insertSearchMessage, searchMessageRows } from './session-search-message-rows' -import { discardSearchBatch, retireSearchSession } from './session-search-write-recovery' +import { discardSearchBatch, retireSearchSession } from './session-search-pending-deletes' import { redactSessionSearchText } from './session-search-redaction' import { sessionSearchPathKey } from './session-search-path-key' export { chunkMessageText } from './session-search-message-rows' export const SEARCH_WRITE_ROWS_PER_STEP = 128 export const SEARCH_WRITE_CHARS_PER_STEP = 256 * 1024 +// Why sampled: the checkpoint costs more than the step it guards, and the backlog it +// watches only grows while a second connection pins a snapshot, which takes seconds. +const WAL_BUDGET_EVERY_STEPS = 16 + +export type SessionSearchApplyOptions = { + /** False once this write is superseded; staging stops without publishing. */ + active?: () => boolean + yieldStep?: () => Promise + /** False once the database is gone; gates the discard tombstone. Defaults to `active`. */ + available?: () => boolean +} + +type ResolvedApplyOptions = Required export type SessionSearchMetadata = Pick< AiVaultSession, @@ -38,7 +51,10 @@ export class SessionSearchIndexWriter { private activePath: string | null = null private invalidated = false private pending: Promise = Promise.resolve() - constructor(private readonly db: SyncDatabase) { + constructor( + private readonly db: SyncDatabase, + private readonly walBudgetBytes: number = SEARCH_WAL_PENDING_BYTES + ) { this.records = new SessionSearchFileRecords(db) } indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null { @@ -84,17 +100,21 @@ export class SessionSearchIndexWriter { apply( update: SessionSearchIndexWrite, - active: () => boolean = () => true, - yieldStep: () => Promise = yieldToEventLoop, - available: () => boolean = active + options: SessionSearchApplyOptions = {} ): Promise { + const active = options.active ?? (() => true) + const resolved: ResolvedApplyOptions = { + active, + yieldStep: options.yieldStep ?? yieldToEventLoop, + available: options.available ?? active + } const run = this.pending .catch(() => undefined) .then(async () => { this.activePath = update.candidate.file.path this.invalidated = false try { - return await this.stage(update, active, yieldStep, available) + return await this.stage(update, resolved) } finally { this.activePath = null } @@ -130,14 +150,11 @@ export class SessionSearchIndexWriter { private async stage( update: SessionSearchIndexWrite, - active: () => boolean, - yieldStep: () => Promise, - available: () => boolean + { active, yieldStep, available }: ResolvedApplyOptions ): Promise { if (!active()) { return false } - assertSearchWalBudget(this.db) const path = update.candidate.file.path const existing = this.file(path) const append = @@ -183,6 +200,7 @@ export class SessionSearchIndexWriter { } const rows = capturedRows() let next = await rows.next() + let step = 0 while (!next.done) { const batch: SessionSearchCapturedMessage[] = [] let chars = 0 @@ -199,7 +217,9 @@ export class SessionSearchIndexWriter { await rows.return(undefined) return false } - assertSearchWalBudget(this.db) + if (step++ % WAL_BUDGET_EVERY_STEPS === 0) { + assertSearchWalBudget(this.db, this.walBudgetBytes) + } this.db.exec('BEGIN IMMEDIATE') try { for (const message of batch) { @@ -212,7 +232,7 @@ export class SessionSearchIndexWriter { } await yieldStep() } - const result = 'result' in update ? await update.result : update + const result = await update.result if (!active() || !unchanged()) { return false } @@ -231,7 +251,10 @@ export class SessionSearchIndexWriter { retireSearchSession(this.db, existing.session_row_id) } this.db.prepare('UPDATE sessions SET index_ready=1 WHERE id=?').run(sessionId) - this.db.prepare('UPDATE search_write_batches SET published=1 WHERE id=?').run(batchId) + // Clearing the pointer before dropping the batch is what makes a recycled + // rowid harmless: no published row can name a later in-flight batch. + this.db.prepare('UPDATE messages SET batch_id=NULL WHERE batch_id=?').run(batchId) + this.db.prepare('DELETE FROM search_write_batches WHERE id=?').run(batchId) this.records.upsertFile(update.candidate, result.byteOffset, sessionId) this.db.exec('COMMIT') } catch (error) { @@ -240,13 +263,12 @@ export class SessionSearchIndexWriter { } return true } finally { - if (available()) { - const batch = this.db - .prepare('SELECT published FROM search_write_batches WHERE id=?') - .get(batchId) as { published: number } | undefined - if (batch?.published === 0) { - discardSearchBatch(this.db, sessionId, batchId, !append) - } + // A surviving batch row means publish never ran, whatever ended the stage. + if ( + available() && + this.db.prepare('SELECT 1 FROM search_write_batches WHERE id=?').get(batchId) + ) { + discardSearchBatch(this.db, sessionId, batchId, !append) } } } diff --git a/src/main/ai-vault-search/session-search-indexing-progress.test.ts b/src/main/ai-vault-search/session-search-indexing-progress.test.ts index cdc364b7c75..06ac0ddb289 100644 --- a/src/main/ai-vault-search/session-search-indexing-progress.test.ts +++ b/src/main/ai-vault-search/session-search-indexing-progress.test.ts @@ -41,6 +41,24 @@ it('groups adjacent live writes into one burst and keeps failures visible', () = expect(progress.snapshot()).toMatchObject({ phase: 'updating', filesTotal: 2, startedAt }) progress.writeFailed() second() + expect(progress.snapshot()).toMatchObject({ phase: 'error', failures: 1 }) +}) + +it('lets a later successful write supersede a write error, but not a failed backfill', () => { + const progress = new SessionSearchIndexingProgress() + const failing = progress.beginWrite() + progress.writeFailed() + failing() + expect(progress.snapshot().phase).toBe('error') + progress.beginWrite()() + expect(progress.snapshot().phase).toBe('complete') + + // A failed backfill stays red: configure() reads this phase to decide whether + // to drop the memoized pass and enumerate again. + progress.discover() + progress.discovered(1, 0) + progress.processed(true) + progress.finish() expect(progress.snapshot().phase).toBe('error') progress.beginWrite()() expect(progress.snapshot().phase).toBe('error') diff --git a/src/main/ai-vault-search/session-search-indexing-progress.ts b/src/main/ai-vault-search/session-search-indexing-progress.ts index dac2d181a72..f086df70d60 100644 --- a/src/main/ai-vault-search/session-search-indexing-progress.ts +++ b/src/main/ai-vault-search/session-search-indexing-progress.ts @@ -13,6 +13,9 @@ export class SessionSearchIndexingProgress { private backfilling = false private activeWrites = 0 private lastWriteAt = 0 + // Why: a failed backfill keeps the badge red until it is retried, but a + // transient write failure must not outlive the next successful write. + private failedWrite = false snapshot(): AiVaultSearchIndexingProgress { return { ...this.value, ...(this.paused ? { phase: 'paused' as const } : {}) } @@ -24,6 +27,7 @@ export class SessionSearchIndexingProgress { discover(): void { this.backfilling = true + this.failedWrite = false this.value = { phase: 'discovering', filesProcessed: 0, @@ -51,7 +55,7 @@ export class SessionSearchIndexingProgress { beginWrite(): () => void { this.activeWrites++ - if (!this.backfilling && !this.paused && this.value.phase !== 'error') { + if (!this.backfilling && !this.paused && (this.value.phase !== 'error' || this.failedWrite)) { if (Date.now() - this.lastWriteAt > 1000 && this.activeWrites === 1) { this.value = { phase: 'updating', @@ -72,6 +76,7 @@ export class SessionSearchIndexingProgress { this.value.filesProcessed++ if (this.activeWrites === 0) { this.value.phase = 'complete' + this.failedWrite = false } } } @@ -79,6 +84,7 @@ export class SessionSearchIndexingProgress { writeFailed(): void { if (!this.backfilling) { + this.failedWrite = true this.value.failures++ this.value.phase = 'error' } diff --git a/src/main/ai-vault-search/session-search-maintenance.ts b/src/main/ai-vault-search/session-search-maintenance.ts deleted file mode 100644 index 9c5cd6e59ae..00000000000 --- a/src/main/ai-vault-search/session-search-maintenance.ts +++ /dev/null @@ -1,81 +0,0 @@ -import { setImmediate as yieldToEventLoop } from 'node:timers/promises' -import type SyncDatabase from '../sqlite/sync-database' -import { assertSearchWalBudget } from './session-search-wal-budget' -import { redactSessionSearchText } from './session-search-redaction' - -const COMPACT_PAGES_PER_STEP = 2000 -const WARM_ROWS_PER_STEP = 50_000 -const SEARCH_LOG_LIMIT = 5000 - -export class SessionSearchMaintenance { - private warmed: Promise | null = null - constructor( - private readonly db: SyncDatabase, - private readonly closed: () => boolean, - private readonly onError: (error: unknown) => void - ) {} - async compact(signal?: AbortSignal): Promise { - try { - let freed = Number(this.db.pragma('freelist_count', { simple: true })) - while (!this.closed() && !signal?.aborted && freed > 0) { - assertSearchWalBudget(this.db) - this.db.pragma(`incremental_vacuum(${COMPACT_PAGES_PER_STEP})`) - const remaining = Number(this.db.pragma('freelist_count', { simple: true })) - // Why: without auto_vacuum the step is a no-op; never spin on it. - if (remaining >= freed) { - return - } - freed = remaining - await yieldToEventLoop() - } - } catch (error) { - this.onError(error) - } - } - - /** - * Reads the messages table through in slices so its pages sit in the OS - * cache before the first query joins against it. Measured on a 4 GB index: - * the first query after a cold start drops from ~1.3 s to ~0.45 s, and each - * slice holds the connection for under 50 ms. - */ - warm(): Promise { - this.warmed ??= this.readMessagesThrough().catch((error) => this.onError(error)) - return this.warmed - } - - private async readMessagesThrough(): Promise { - const max = ( - this.db.prepare('SELECT max(id) AS id FROM messages').get() as { id: number | null } - ).id - const touch = this.db.prepare( - 'SELECT count(*) FROM messages WHERE id BETWEEN ? AND ? AND role IS NOT NULL' - ) - for (let low = 1; max !== null && low <= max; low += WARM_ROWS_PER_STEP) { - if (this.closed()) { - return - } - touch.get(low, low + WARM_ROWS_PER_STEP - 1) - await yieldToEventLoop() - } - } - - logQuery(query: string, route: string, hits: number, durationMs: number): void { - try { - assertSearchWalBudget(this.db) - this.db - .prepare( - 'INSERT INTO search_log(ts, query, route, hits, duration_ms) VALUES (?, ?, ?, ?, ?)' - ) - .run(new Date().toISOString(), redactSessionSearchText(query), route, hits, durationMs) - this.db - .prepare( - `DELETE FROM search_log WHERE id <= ( - SELECT id FROM search_log ORDER BY id DESC LIMIT 1 OFFSET ?)` - ) - .run(SEARCH_LOG_LIMIT) - } catch (error) { - this.onError(error) - } - } -} diff --git a/src/main/ai-vault-search/session-search-opencode-cancellation.test.ts b/src/main/ai-vault-search/session-search-opencode-cancellation.test.ts index 43ceea0e1dc..c5f2d566e62 100644 --- a/src/main/ai-vault-search/session-search-opencode-cancellation.test.ts +++ b/src/main/ai-vault-search/session-search-opencode-cancellation.test.ts @@ -82,7 +82,7 @@ class LoopbackWorker extends EventEmitter { ) .then(async (session) => { await capture.flush() - this.emit('message', { id: request.id, ok: true, value: { session } }) + this.emit('message', { id: request.id, kind: 'result', value: { session } }) }) .catch((error) => { if (!this.terminated) { diff --git a/src/main/ai-vault-search/session-search-opencode-freshness.test.ts b/src/main/ai-vault-search/session-search-opencode-freshness.test.ts index e2cb107aba6..07c85047947 100644 --- a/src/main/ai-vault-search/session-search-opencode-freshness.test.ts +++ b/src/main/ai-vault-search/session-search-opencode-freshness.test.ts @@ -185,6 +185,26 @@ describe('OpenCode SQLite session freshness', () => { expect(store.coverage().messagesIndexed).toBe(2) }) + it('keeps listing a session whose part blob the capture read cannot decode', async () => { + const dbPath = await createOpenCodeDb() + const db = new Database(dbPath) + // Valid JSON that is not an object, so SQLite's json_extract tolerates it in + // the preview query and only the search capture read has to survive it. + db.prepare( + `INSERT INTO part (id, message_id, session_id, time_created, time_updated, data) + VALUES ('prt_bad', 'msg_1', ?, ?, ?, 'null')` + ).run(SESSION_ID, CREATED_MS + 600, CREATED_MS + 600) + db.close() + + // The parse must not reject: an index failure may never cost the session its + // place in the list, and the readable turns must still reach the index. + await expect(refreshRecent(dbPath)).resolves.toMatchObject({ fullParses: 1 }) + expect(store.search({ query: 'vacuum quota' }).hits).toMatchObject([ + { agent: 'opencode', sessionId: SESSION_ID } + ]) + expect(store.coverage().messagesIndexed).toBe(1) + }) + it('reuses the cached parse and leaves the index alone when nothing changed', async () => { const dbPath = await createOpenCodeDb() await refreshRecent(dbPath) diff --git a/src/main/ai-vault-search/session-search-page-warmup.ts b/src/main/ai-vault-search/session-search-page-warmup.ts new file mode 100644 index 00000000000..e4be5d63cfd --- /dev/null +++ b/src/main/ai-vault-search/session-search-page-warmup.ts @@ -0,0 +1,27 @@ +import { setImmediate as yieldToEventLoop } from 'node:timers/promises' +import type SyncDatabase from '../sqlite/sync-database' + +const WARM_ROWS_PER_STEP = 50_000 + +/** + * Reads the messages table through in slices so its pages sit in the OS cache + * before the first query joins against it. Measured on a 4 GB index: the first + * query after a cold start drops from ~1.3 s to ~0.45 s, and each slice holds + * the connection for under 50 ms. + */ +export async function warmSessionSearchPages( + db: SyncDatabase, + stopped: () => boolean +): Promise { + const max = (db.prepare('SELECT max(id) AS id FROM messages').get() as { id: number | null }).id + const touch = db.prepare( + 'SELECT count(*) FROM messages WHERE id BETWEEN ? AND ? AND role IS NOT NULL' + ) + for (let low = 1; max !== null && low <= max; low += WARM_ROWS_PER_STEP) { + if (stopped()) { + return + } + touch.get(low, low + WARM_ROWS_PER_STEP - 1) + await yieldToEventLoop() + } +} diff --git a/src/main/ai-vault-search/session-search-parse-candidates.ts b/src/main/ai-vault-search/session-search-parse-candidates.ts index d128dabf87b..e6ba99f7daa 100644 --- a/src/main/ai-vault-search/session-search-parse-candidates.ts +++ b/src/main/ai-vault-search/session-search-parse-candidates.ts @@ -16,11 +16,16 @@ import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' import type { SessionSearchStore } from './session-search-store' import { pauseBackfill } from './session-search-backfill-pacing' +export type ParseSearchCandidatesOptions = { + signal?: AbortSignal + /** Backfill only: report each file to the progress bar and yield to waiting searches. */ + onFileProcessed?: (failed: boolean) => Promise +} + export async function parseSearchCandidates( store: SessionSearchStore, candidates: SessionFileCandidate[], - signal?: AbortSignal, - waitForSearches?: () => Promise + { signal, onFileProcessed }: ParseSearchCandidatesOptions = {} ): Promise { const stats = createSessionParseStats() let sinceYield = 0 @@ -75,9 +80,8 @@ export async function parseSearchCandidates( error instanceof Error ? error.name : 'ParseError' ) } - if (waitForSearches && !signal?.aborted) { - store.indexing.processed(failed || store.failures > failures) - await waitForSearches() + if (onFileProcessed && !signal?.aborted) { + await onFileProcessed(failed || store.failures > failures) } sinceYield++ if (sinceYield >= 8) { diff --git a/src/main/ai-vault-search/session-search-pause.test.ts b/src/main/ai-vault-search/session-search-pause.test.ts index 00d7f1e3457..74ec3800419 100644 --- a/src/main/ai-vault-search/session-search-pause.test.ts +++ b/src/main/ai-vault-search/session-search-pause.test.ts @@ -36,7 +36,7 @@ beforeEach(async () => { }) afterEach(async () => { - service.dispose() + await service.close() await rm(root, { recursive: true, force: true }) }) @@ -65,7 +65,7 @@ it('pauses ordinary scanner writes and query refreshes, then catches up without it('keeps pause across a scanner restart and preserves saved results', async () => { await service.ensureBackfill(roots) await service.configure({ ...enabled, paused: true }, roots) - service.dispose() + await service.close() service = new SessionSearchService({ databasePath, ...enabled, paused: true }) await appendFile(file, `${userRecord(1, 'restartneedle')}\n`) await service.ensureBackfill(roots) @@ -94,7 +94,7 @@ it('clear while paused stays paused; disabling closes the sink and retains the p it('an interrupted write cannot be revived by a quick resume', async () => { const { SessionSearchStore } = await import('./session-search-store') - const { stagedWriteUpdate } = await import('./session-search-staged-write-fixtures') + const { stagedWriteUpdate } = await import('./session-search-staged-write-test-fixture') const store = getSessionSearchIndexSink() as InstanceType await store.apply(stagedWriteUpdate('savedneedle', 1)) let entered!: () => void diff --git a/src/main/ai-vault-search/session-search-write-recovery.ts b/src/main/ai-vault-search/session-search-pending-deletes.ts similarity index 63% rename from src/main/ai-vault-search/session-search-write-recovery.ts rename to src/main/ai-vault-search/session-search-pending-deletes.ts index 6e658eef620..69de9f9f603 100644 --- a/src/main/ai-vault-search/session-search-write-recovery.ts +++ b/src/main/ai-vault-search/session-search-pending-deletes.ts @@ -1,12 +1,10 @@ import type SyncDatabase from '../sqlite/sync-database' -export function retireSearchSession( - db: SyncDatabase, - sessionId: number, - path = `\0session:${sessionId}` -): void { +// Why the NUL prefix: tombstones share the file-path key space with real files, and +// `\0` cannot occur in one, so a synthetic key still gets the per-path cleanup mutex. +export function retireSearchSession(db: SyncDatabase, sessionId: number): void { db.prepare('INSERT OR IGNORE INTO search_pending_deletes(path,session_row_id) VALUES (?,?)').run( - path, + `\0session:${sessionId}`, sessionId ) } @@ -28,9 +26,10 @@ export function discardSearchBatch( /** Only called on open, before this store can have active writers. */ export function recoverSearchWrites(db: SyncDatabase): void { + // A staging session or a batch row that outlived its writer is by definition unfinished: + // publish clears both in the same transaction that makes the rows visible. db.exec(`INSERT OR IGNORE INTO search_pending_deletes(path,session_row_id) SELECT char(0)||'session:'||id,id FROM sessions WHERE index_ready=0; INSERT OR IGNORE INTO search_pending_deletes(path,session_row_id,batch_id) - SELECT char(0)||'batch:'||b.id,b.session_row_id,b.id FROM search_write_batches b - JOIN sessions s ON s.id=b.session_row_id WHERE b.published=0 AND s.index_ready=1`) + SELECT char(0)||'batch:'||id,session_row_id,id FROM search_write_batches`) } diff --git a/src/main/ai-vault-search/session-search-query-log.ts b/src/main/ai-vault-search/session-search-query-log.ts new file mode 100644 index 00000000000..30f0324cd49 --- /dev/null +++ b/src/main/ai-vault-search/session-search-query-log.ts @@ -0,0 +1,24 @@ +import type SyncDatabase from '../sqlite/sync-database' +import { redactSessionSearchText } from './session-search-redaction' + +const SEARCH_LOG_LIMIT = 5000 + +/** Telemetry the eval set is rebuilt from; the query text is redacted before it lands. */ +export function logSessionSearchQuery( + db: SyncDatabase, + entry: { query: string; route: string; hits: number; durationMs: number } +): void { + db.prepare( + 'INSERT INTO search_log(ts, query, route, hits, duration_ms) VALUES (?, ?, ?, ?, ?)' + ).run( + new Date().toISOString(), + redactSessionSearchText(entry.query), + entry.route, + entry.hits, + entry.durationMs + ) + db.prepare( + `DELETE FROM search_log WHERE id <= ( + SELECT id FROM search_log ORDER BY id DESC LIMIT 1 OFFSET ?)` + ).run(SEARCH_LOG_LIMIT) +} diff --git a/src/main/ai-vault-search/session-search-query-operators.test.ts b/src/main/ai-vault-search/session-search-query-operators.test.ts index 8752b969c67..96947664f27 100644 --- a/src/main/ai-vault-search/session-search-query-operators.test.ts +++ b/src/main/ai-vault-search/session-search-query-operators.test.ts @@ -10,6 +10,7 @@ import { parseTranscript, userRecord } from './session-search-transcript-fixture const APP_ID = 'aaaaaaaa-0000-4000-8000-00000000000a' const SERVICE_ID = 'aaaaaaaa-0000-4000-8000-00000000000b' const NEWER_APP_ID = 'aaaaaaaa-0000-4000-8000-00000000000c' +const ACCENTED_ID = 'aaaaaaaa-0000-4000-8000-00000000000d' let tempRoots: string[] = [] let root: string @@ -93,6 +94,30 @@ describe('repo: and path: operators at the query level', () => { expect(sessionIds('repo:service', { scopePaths: ['/repo'] })).toEqual([]) }) + it('ORs terms within one operator key and ANDs across keys', () => { + expect(sessionIds('harbor repo:app repo:service').sort()).toEqual( + [APP_ID, NEWER_APP_ID, SERVICE_ID].sort() + ) + expect(sessionIds('harbor path:/repo path:/other').sort()).toEqual( + [APP_ID, NEWER_APP_ID, SERVICE_ID].sort() + ) + // Two absolute paths would be unsatisfiable if same-key terms ANDed. + expect(sessionIds('harbor path:/repo/app path:/other').sort()).toEqual( + [APP_ID, NEWER_APP_ID, SERVICE_ID].sort() + ) + expect(sessionIds('harbor repo:app path:/other')).toEqual([]) + }) + + it('folds a substring path term as far as SQLite can, and an absolute one not at all', async () => { + // LIKE folds ASCII on both sides; `lower()` would fold ASCII and still miss `É`. + await indexSession(ACCENTED_ID, '/repo/CAFÉ', 'harbor lantern') + expect(sessionIds('harbor path:café')).toEqual([]) + expect(sessionIds('harbor path:CAFÉ')).toEqual([ACCENTED_ID]) + expect(sessionIds('harbor path:APP').sort()).toEqual([APP_ID, NEWER_APP_ID].sort()) + // An absolute term is an identity claim, so POSIX case is not folded. + expect(sessionIds('harbor path:/REPO/app')).toEqual([]) + }) + it('keeps operator text out of the FTS expression', () => { // `harbo` is one edit from the indexed `harbor`: if the operator value // reached the planner it would come back on a typo+ route. diff --git a/src/main/ai-vault-search/session-search-query-planner.ts b/src/main/ai-vault-search/session-search-query-planner.ts index 690274d1e47..965dea17481 100644 --- a/src/main/ai-vault-search/session-search-query-planner.ts +++ b/src/main/ai-vault-search/session-search-query-planner.ts @@ -32,13 +32,20 @@ export function isLiteralQuery(query: string): boolean { return QUOTED.test(query) || LITERAL_SHAPE.test(query) } -function indexTokens(query: string): string[] { +/** + * The tokenizer contract, unfolded: the same boundaries FTS5 draws for + * `unicode61 tokenchars '_.-/+'`. Pinned against real `fts5vocab` output in + * session-search-fts5-contract.test.ts, which is what makes it safe to plan a + * query without asking SQLite. + */ +export function indexTokens(query: string, limit = Number.POSITIVE_INFINITY): string[] { const out: string[] = [] for (const match of query.matchAll(INDEX_TOKEN)) { const token = match[0] + // Separators alone (`--`, `...`) are a token to FTS5 but never a search term. if (/[\p{L}\p{N}\p{Co}]/u.test(token)) { out.push(token) - if (out.length >= MAX_BODY_TERMS) { + if (out.length >= limit) { break } } @@ -47,7 +54,7 @@ function indexTokens(query: string): string[] { } export function planSessionSearchQuery(query: string): SessionSearchQueryPlan { - const raw = indexTokens(query) + const raw = indexTokens(query, MAX_BODY_TERMS) let body = isLiteralQuery(query) ? raw : raw.filter((token) => !STOP_WORDS.has(token.toLowerCase())) diff --git a/src/main/ai-vault-search/session-search-query-regressions.test.ts b/src/main/ai-vault-search/session-search-query-regressions.test.ts index f13b5212f46..a7c4ace6f3e 100644 --- a/src/main/ai-vault-search/session-search-query-regressions.test.ts +++ b/src/main/ai-vault-search/session-search-query-regressions.test.ts @@ -1,78 +1,93 @@ import { removeTree } from '../../shared/windows-transient-lock-removal' import { sessionSearchPathKey } from './session-search-path-key' import { describe, it, expect } from 'vitest' +import type SyncDatabase from '../sqlite/sync-database' import { SessionSearchStore } from './session-search-store' import { SessionSearchService } from './session-search-service' +import { openSessionSearchIndexFile } from './session-search-staged-write-test-fixture' import { isolatedScanRoots } from '../ai-vault/session-scanner-test-fixtures' import { mkdir, mkdtemp, writeFile, utimes } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { userRecord, parseTranscript } from './session-search-transcript-fixtures' import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' -function add(store: SessionSearchStore, id: number, cwd: string, text: string, count = 1) { - store.db - .prepare( - `INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,message_count,resume_command) VALUES (?, 'claude', ?, ?, 'audit fixture', ?, ?, 1, '')` - ) - .run(id, String(id), `/synthetic/${id}`, cwd, sessionSearchPathKey(cwd)) - for (let n = 0; n < count; n++) { - const row = Number( - store.db.prepare("INSERT INTO messages(session_row_id,role) VALUES (?, 'user')").run(id) - .lastInsertRowid - ) - store.db.prepare('INSERT INTO messages_fts(rowid,user_text) VALUES (?,?)').run(row, text) + +/** The store keeps its connection private, so synthetic rows go in through a second one. */ +async function withIndex( + run: (db: SyncDatabase, store: SessionSearchStore) => void +): Promise { + const index = await openSessionSearchIndexFile('ss-query-regressions') + const store = new SessionSearchStore(index.path) + try { + run(index.db, store) + } finally { + store.close() + await index.close() } } + +function add( + db: SyncDatabase, + id: number, + cwd: string, + text: string, + count = 1, + transcriptPath?: string +) { + db.prepare( + `INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,message_count,resume_command) VALUES (?, 'claude', ?, ?, 'audit fixture', ?, ?, 1, '')` + ).run(id, String(id), `/synthetic/${id}`, cwd, sessionSearchPathKey(cwd, transcriptPath)) + for (let n = 0; n < count; n++) { + const row = Number( + db.prepare("INSERT INTO messages(session_row_id,role) VALUES (?, 'user')").run(id) + .lastInsertRowid + ) + db.prepare('INSERT INTO messages_fts(rowid,user_text) VALUES (?,?)').run(row, text) + } +} + describe('search correctness regressions', () => { - it('finds a scoped match behind 600 out-of-scope rows', () => { - const s = new SessionSearchStore(':memory:') - try { - add(s, 1, '/unrelated', 'auditneedle', 600) - add(s, 2, '/target', 'auditneedle padding') + it('finds a scoped match behind 600 out-of-scope rows', async () => { + await withIndex((db, store) => { + add(db, 1, '/unrelated', 'auditneedle', 600) + add(db, 2, '/target', 'auditneedle padding') expect( - s.search({ query: 'auditneedle', scopePaths: ['/target'] }).hits.map((h) => h.sessionId) + store.search({ query: 'auditneedle', scopePaths: ['/target'] }).hits.map((h) => h.sessionId) ).toEqual(['2']) - } finally { - s.close() - } + }) }) - it('falls back when the exact hit is out of scope', () => { - const s = new SessionSearchStore(':memory:') - try { - add(s, 1, '/unrelated', 'resolveTerminalPath') - add(s, 2, '/target', 'resolve terminal path') + + it('falls back when the exact hit is out of scope', async () => { + await withIndex((db, store) => { + add(db, 1, '/unrelated', 'resolveTerminalPath') + add(db, 2, '/target', 'resolve terminal path') expect( - s + store .search({ query: 'resolveTerminalPath', scopePaths: ['/target'] }) .hits.map((h) => h.sessionId) ).toEqual(['2']) - } finally { - s.close() - } + }) }) - it('phrase route requires adjacent ordered tokens', () => { - const s = new SessionSearchStore(':memory:') - try { - add(s, 1, '/target', 'beta separated alpha') - expect(s.search({ query: '"alpha beta"' }).route).toBe('and') - } finally { - s.close() - } + + it('phrase route requires adjacent ordered tokens', async () => { + await withIndex((db, store) => { + add(db, 1, '/target', 'beta separated alpha') + expect(store.search({ query: '"alpha beta"' }).route).toBe('and') + }) }) - it('unicode term indexed by FTS is searchable', () => { - const s = new SessionSearchStore(':memory:') - try { - add(s, 1, '/target', '안녕하세요') + + it('unicode term indexed by FTS is searchable', async () => { + await withIndex((db, store) => { + add(db, 1, '/target', '안녕하세요') expect( - s.db + db .prepare('SELECT count(*) as n FROM messages_fts WHERE messages_fts MATCH ?') .get('안녕하세요') ).toEqual({ n: 1 }) - expect(s.search({ query: '안녕하세요' }).hits).toHaveLength(1) - } finally { - s.close() - } + expect(store.search({ query: '안녕하세요' }).hits).toHaveLength(1) + }) }) + it('ordinary list parse respects selected history retention', async () => { resetSessionParseCacheForTests() const root = await mkdtemp(join(tmpdir(), 'orca-audit-retention-')) @@ -93,7 +108,7 @@ describe('search correctness regressions', () => { await parseTranscript(path) expect(s.coverage().sessionsIndexed).toBe(0) } finally { - s.dispose() + await s.close() await removeTree(root) } }) @@ -104,19 +119,25 @@ describe('path and retrieval contracts', () => { ['C:\\Work\\App', 'c:/work/app', true], ['C:\\Work\\App\\src', 'c:/work/app', true], ['/work/APP/src', '/work/app', false], - ['/work/café', '/work/cafe\u0301', true], + ['/work/caf\u00e9', '/work/cafe\u0301', true], ['/work/app-other', '/work/app', false], ['/work/a_b/src', '/work/a_b', true], - ['/work/axb/src', '/work/a_b', false] - ])('scopes %s under %s: %s', (cwd, scope, expected) => { - const store = new SessionSearchStore(':memory:') - try { - add(store, 1, cwd, 'needle') - expect(store.search({ query: 'needle', scopePaths: [scope] }).hits.length > 0).toBe(expected) + ['/work/axb/src', '/work/a_b', false], + // Roots: both key to a degenerate prefix ('' and 'c:'), which is where a + // range bound is easiest to get wrong. A Windows key is not under POSIX '/'. + ['/', '/', true], + ['/work/app', '/', true], + ['C:\\Work\\App', '/', false], + ['C:\\', 'C:\\', true], + ['C:\\Work\\App', 'C:\\', true] + ])('scopes %s under %s: %s', async (cwd, scope, expected) => { + await withIndex((db, store) => { + add(db, 1, cwd as string, 'needle') + expect(store.search({ query: 'needle', scopePaths: [scope as string] }).hits.length > 0).toBe( + expected + ) expect(store.search({ query: `needle path:"${scope}"` }).hits.length > 0).toBe(expected) - } finally { - store.close() - } + }) }) it('keeps WSL distro identity and Linux path case', () => { @@ -128,28 +149,53 @@ describe('path and retrieval contracts', () => { ) }) - it('newest returns distinct sessions even when one has over 600 matching rows', () => { - const store = new SessionSearchStore(':memory:') - try { - add(store, 1, '/app', 'needle', 650) - add(store, 2, '/app', 'needle padding') - store.db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-06', 1) - store.db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-05', 2) - expect( - store.search({ query: 'needle', sort: 'newest' }).hits.map((hit) => hit.sessionId) - ).toEqual(['1', '2']) - } finally { - store.close() + // Both sorts collapse to one row per session before the candidate limit, so + // the 650-row session cannot crowd the one-row session off the page. + it.each(['relevance', 'newest'] as const)( + '%s returns distinct sessions even when one has over 600 matching rows', + async (sort) => { + await withIndex((db, store) => { + add(db, 1, '/app', 'needle', 650) + add(db, 2, '/app', 'needle padding') + db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-06', 1) + db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-05', 2) + expect(store.search({ query: 'needle', sort }).hits.map((hit) => hit.sessionId)).toEqual([ + '1', + '2' + ]) + }) } + ) + + // The index writer can prove a WSL session's distro from its transcript path; + // a query-time term never can. Orca stores a WSL workspace as the UNC path, + // which is the spelling that keys the same way the writer did. + it('scopes a WSL session by its UNC workspace path, not by the Linux spelling', async () => { + await withIndex((db, store) => { + add( + db, + 1, + '/home/ada/app', + 'needle', + 1, + '\\\\wsl.localhost\\Ubuntu\\home\\ada\\session.jsonl' + ) + const hits = (scope: string): number => + store.search({ query: 'needle', scopePaths: [scope] }).hits.length + expect(hits('\\\\wsl$\\Ubuntu\\home\\ada\\app')).toBe(1) + expect(hits('\\\\wsl.localhost\\Ubuntu\\home\\ada')).toBe(1) + expect(hits('/home/ada/app')).toBe(0) + expect(hits('\\\\wsl$\\Debian\\home\\ada\\app')).toBe(0) + }) }) - it.each(['café', 'C', 'R', 'x', '修复'])('searches unicode61 token %s', (text) => { - const store = new SessionSearchStore(':memory:') - try { - add(store, 1, '/app', text) - expect(store.search({ query: text }).hits).toHaveLength(1) - } finally { - store.close() + it.each(['caf\u00e9', 'C', 'R', 'x', '\u4fee\u590d'])( + 'searches unicode61 token %s', + async (text) => { + await withIndex((db, store) => { + add(db, 1, '/app', text) + expect(store.search({ query: text }).hits).toHaveLength(1) + }) } - }) + ) }) diff --git a/src/main/ai-vault-search/session-search-query.ts b/src/main/ai-vault-search/session-search-query.ts index 218fea45b96..e69b8b932a1 100644 --- a/src/main/ai-vault-search/session-search-query.ts +++ b/src/main/ai-vault-search/session-search-query.ts @@ -5,12 +5,16 @@ import type { AiVaultSearchRoute } from '../../shared/ai-vault-search-types' import { - AI_VAULT_SEARCH_LIMIT_DEFAULT, - AI_VAULT_SEARCH_LIMIT_MAX, AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE, AI_VAULT_SEARCH_SNIPPET_MARK_OPEN } from '../../shared/ai-vault-search-types' -import { sessionFields, type SessionRow } from './session-search-session-row' +import { + rankSessionHits, + resolveLimit, + sessionFields, + type MessageRow, + type SessionRow +} from './session-search-hit-ranking' import { andExpression, orExpression, @@ -18,7 +22,7 @@ import { planSessionSearchQuery, type SessionSearchQueryPlan } from './session-search-query-planner' -import { isCollapsibleContentHash } from './session-search-content-hash' +import { VISIBLE_MESSAGES, VISIBLE_SESSIONS } from './session-search-schema' import { SessionSearchTypoRepair } from './session-search-typo-repair' import { sessionRowFilter, type SessionRowFilter } from './session-search-row-filter' import { @@ -31,29 +35,12 @@ const FULL_WEIGHTS = '3.0, 2.0, 1.0, 1.0' const CONVERSATION_WEIGHTS = '3.0, 2.0' // Candidate messages fetched before rolling up to sessions; more does not help. const MESSAGE_CANDIDATE_LIMIT = 600 -// Subtracted per session: `0.02 · ln(1 + messages)`; slightly positive on both eval sets. -const LENGTH_PRIOR = 0.02 const SNIPPET_TOKENS = 12 // Why: single brackets are everywhere in code transcripts (`arr[0]`, regex // classes, markdown links) and would read as matches; doubled ones are rare. const SNIPPET_MARK_OPEN = AI_VAULT_SEARCH_SNIPPET_MARK_OPEN const SNIPPET_MARK_CLOSE = AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE -type MessageRow = { - rowid: number - score: number - session_row_id: number - role: string - ts: string | null -} - -type ScoredSession = { - session: SessionRow - message: MessageRow - score: number - duplicateCount: number -} - /** One search pass: the caller's args plus everything the operators decided. */ type Retrieval = { args: AiVaultSearchArgs @@ -81,15 +68,9 @@ export class SessionSearchQuery { const retrieval: Retrieval = { args, tier: args.tier ?? 'full', - filter: sessionRowFilter(args, split), + filter: sessionRowFilter(args, split, cutoffMs), text: split.text } - if (cutoffMs !== null) { - retrieval.filter.conditions.push( - 'id IN (SELECT session_row_id FROM files WHERE mtime_ms >= ?)' - ) - retrieval.filter.values.push(String(cutoffMs)) - } const plan = planSessionSearchQuery(retrieval.text) if (plan.terms.length === 0) { // Operators with no free text still name a scope, so answer with the @@ -124,7 +105,7 @@ export class SessionSearchQuery { return ( this.db .prepare( - `SELECT * FROM sessions ${where} ORDER BY updated_at DESC LIMIT ${resolveLimit(retrieval.args)}` + `SELECT * FROM ${VISIBLE_SESSIONS} ${where} ORDER BY updated_at DESC LIMIT ${resolveLimit(retrieval.args)}` ) .all(...values) as SessionRow[] ).map((session) => ({ @@ -172,22 +153,21 @@ export class SessionSearchQuery { private match(expression: string, retrieval: Retrieval): MessageRow[] { const { tier, filter, args } = retrieval const eligible = filter.conditions.length - ? ` AND m.session_row_id IN (SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')})` + ? ` AND m.session_row_id IN (SELECT id FROM ${VISIBLE_SESSIONS} WHERE ${filter.conditions.join(' AND ')})` : '' const table = tier === 'full' ? 'messages_fts' : 'conversation_fts' const weights = tier === 'full' ? FULL_WEIGHTS : CONVERSATION_WEIGHTS const matched = `SELECT ${table}.rowid AS rowid, -bm25(${table}, ${weights}) AS score, m.session_row_id, m.role, m.ts, s.updated_at - FROM ${table} JOIN messages m ON m.id = ${table}.rowid - JOIN sessions s ON s.id = m.session_row_id WHERE ${table} MATCH ?${eligible} - AND (m.batch_id IS NULL OR m.batch_id NOT IN (SELECT id FROM search_write_batches WHERE published=0))` - // Collapse messages before newest ordering so one long session cannot occupy the whole page. - const sql = - args.sort === 'newest' - ? `WITH matched AS MATERIALIZED (${matched}) - SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched - GROUP BY session_row_id ORDER BY updated_at DESC, score DESC LIMIT ${MESSAGE_CANDIDATE_LIMIT}` - : `${matched} ORDER BY score DESC LIMIT ${MESSAGE_CANDIDATE_LIMIT}` + FROM ${table} JOIN ${VISIBLE_MESSAGES} m ON m.id = ${table}.rowid + JOIN ${VISIBLE_SESSIONS} s ON s.id = m.session_row_id WHERE ${table} MATCH ?${eligible}` + // Why: collapse to one row per session BEFORE the candidate limit, on both + // sort orders, so a single long session cannot occupy the whole page. + // `max(score)` makes SQLite pick that session's best row for the bare columns. + const order = args.sort === 'newest' ? 'updated_at DESC, score DESC' : 'score DESC' + const sql = `WITH matched AS MATERIALIZED (${matched}) + SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched + GROUP BY session_row_id ORDER BY ${order} LIMIT ${MESSAGE_CANDIDATE_LIMIT}` return this.db.prepare(sql).all(expression, ...filter.values) as MessageRow[] } @@ -196,46 +176,18 @@ export class SessionSearchQuery { retrieval: Retrieval, plan: SessionSearchQueryPlan ): AiVaultSearchHit[] { - const best = new Map() - for (const row of rows) { - const current = best.get(row.session_row_id) - if (!current || row.score > current.score) { - best.set(row.session_row_id, row) - } - } + // `match` already grouped to one best row per session. + const best = new Map(rows.map((row) => [row.session_row_id, row])) if (best.size === 0) { return [] } - const sessions = this.loadSessions([...best.keys()], retrieval.filter) - const scored = collapseForks( - sessions.map((session) => { - const message = best.get(session.id)! - return { - session, - message, - score: message.score - LENGTH_PRIOR * Math.log(1 + session.message_count), - duplicateCount: 1 - } - }) - ) - scored.sort((left, right) => - retrieval.args.sort === 'newest' - ? (right.session.updated_at ?? '').localeCompare(left.session.updated_at ?? '') - : right.score - left.score - ) const table = retrieval.tier === 'full' ? 'messages_fts' : 'conversation_fts' - return scored - .slice(0, resolveLimit(retrieval.args)) - .map(({ session, message, score, duplicateCount }) => ({ - ...sessionFields(session), - score, - ...(duplicateCount > 1 ? { duplicateCount } : {}), - evidence: { - role: message.role as AiVaultSearchHit['evidence']['role'], - timestamp: message.ts, - snippet: this.snippet(table, message.rowid, plan) - } - })) + return rankSessionHits( + this.loadSessions([...best.keys()], retrieval.filter), + best, + retrieval.args, + (message) => this.snippet(table, message.rowid, plan) + ) } // Why: the snippet must highlight the terms that actually retrieved the row, @@ -274,50 +226,7 @@ export class SessionSearchQuery { private loadSessions(ids: number[], filter: SessionRowFilter): SessionRow[] { const conditions = [`id IN (${ids.map(() => '?').join(',')})`, ...filter.conditions] return this.db - .prepare(`SELECT * FROM sessions WHERE ${conditions.join(' AND ')}`) + .prepare(`SELECT * FROM ${VISIBLE_SESSIONS} WHERE ${conditions.join(' AND ')}`) .all(...ids, ...filter.values) as SessionRow[] } } - -// Why: the desktop IPC forwards its payload unvalidated, so a non-positive -// limit must be clamped here or `LIMIT -1` / `slice(0, -1)` leak through. -function resolveLimit(args: AiVaultSearchArgs): number { - const requested = Number.isInteger(args.limit) - ? (args.limit as number) - : AI_VAULT_SEARCH_LIMIT_DEFAULT - return Math.min(Math.max(1, requested), AI_VAULT_SEARCH_LIMIT_MAX) -} - -/** - * Folds forked copies of one conversation into a single hit: same opening - * prefix, newest `updated_at` wins, the rest become `duplicateCount`. Done here - * and not at write time so index rows stay per file (cursors and deletes). - */ -function collapseForks(scored: ScoredSession[]): ScoredSession[] { - const groups = new Map() - for (const entry of scored) { - const { content_hash: hash, content_hash_count: count, id } = entry.session - const key = isCollapsibleContentHash(hash, count) ? `hash:${hash}` : `session:${id}` - const group = groups.get(key) - if (group) { - group.push(entry) - } else { - groups.set(key, [entry]) - } - } - const collapsed: ScoredSession[] = [] - for (const group of groups.values()) { - if (group.length === 1) { - collapsed.push(group[0]!) - continue - } - const winner = group.reduce((best, entry) => (isNewer(entry, best) ? entry : best)) - collapsed.push({ ...winner, duplicateCount: group.length }) - } - return collapsed -} - -function isNewer(entry: ScoredSession, best: ScoredSession): boolean { - const order = (entry.session.updated_at ?? '').localeCompare(best.session.updated_at ?? '') - return order === 0 ? entry.score > best.score : order > 0 -} diff --git a/src/main/ai-vault-search/session-search-refresh-lane.ts b/src/main/ai-vault-search/session-search-refresh-lane.ts index a4b95af42be..2e85286e008 100644 --- a/src/main/ai-vault-search/session-search-refresh-lane.ts +++ b/src/main/ai-vault-search/session-search-refresh-lane.ts @@ -3,6 +3,7 @@ import { sessionCandidatesFromDiscoveries } from '../ai-vault/session-scanner-ca import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' import { waitForPromiseWithSignal, throwIfSignalAborted } from '../../shared/abort-signal-reason' +import { stableInFlightKey } from '../../shared/in-flight-promise-dedupe' import type { SessionSearchScanRoots } from './session-search-service' type Refresh = { controller: AbortController; promise: Promise; users: number } @@ -18,7 +19,7 @@ export class SessionSearchRefreshLane { signal?: AbortSignal ): Promise { throwIfSignalAborted(signal) - const key = JSON.stringify(Object.entries(roots).sort(([a], [b]) => a.localeCompare(b))) + const key = stableInFlightKey(Object.entries(roots).sort(([a], [b]) => a.localeCompare(b))) let run = this.runs.get(key) if (!run) { const controller = new AbortController() diff --git a/src/main/ai-vault-search/session-search-retention-delete.test.ts b/src/main/ai-vault-search/session-search-retention-delete.test.ts index eaa266d1369..360a511c318 100644 --- a/src/main/ai-vault-search/session-search-retention-delete.test.ts +++ b/src/main/ai-vault-search/session-search-retention-delete.test.ts @@ -1,16 +1,13 @@ -import { removeTree } from '../../shared/windows-transient-lock-removal' -import { mkdtemp } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' import { expect, it } from 'vitest' +import type SyncDatabase from '../sqlite/sync-database' import { SessionSearchStore } from './session-search-store' +import { openSessionSearchIndexFile } from './session-search-staged-write-test-fixture' import { deleteExpiredSearchFiles, RETENTION_DELETE_ROWS_PER_STEP } from './session-search-retention-delete' -function seed(store: SessionSearchStore, id: number, rows: number, mtime: number) { - const db = store.db +function seed(db: SyncDatabase, id: number, rows: number, mtime: number) { db.prepare(`INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,resume_command) VALUES (?, 'claude', ?, ?, 'synthetic retention', '/fixture', '/fixture', '')`).run( id, @@ -37,21 +34,22 @@ function seed(store: SessionSearchStore, id: number, rows: number, mtime: number } it('yields within a large file while hiding partial rows and preserving unrelated sessions', async () => { - const store = new SessionSearchStore(':memory:') - seed(store, 1, 1025, 1) - seed(store, 2, 1, 200) + const index = await openSessionSearchIndexFile('ss-retention-yield') + const store = new SessionSearchStore(index.path) + seed(index.db, 1, 1025, 1) + seed(index.db, 2, 1, 200) let previous = 1025 let steps = 0 try { await deleteExpiredSearchFiles( - store.db, + index.db, 100, () => false, () => {}, async () => { const left = Number( ( - store.db.prepare('SELECT count(*) AS n FROM messages WHERE session_row_id=1').get() as { + index.db.prepare('SELECT count(*) AS n FROM messages WHERE session_row_id=1').get() as { n: number } ).n @@ -66,25 +64,25 @@ it('yields within a large file while hiding partial rows and preserving unrelate } ) expect(steps).toBe(5) - expect(store.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 1 }) - expect(store.db.prepare('SELECT count(*) AS n FROM conversation_fts').get()).toEqual({ n: 1 }) - expect(store.db.prepare('SELECT count(*) AS n FROM search_pending_deletes').get()).toEqual({ + expect(index.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 1 }) + expect(index.db.prepare('SELECT count(*) AS n FROM conversation_fts').get()).toEqual({ n: 1 }) + expect(index.db.prepare('SELECT count(*) AS n FROM search_pending_deletes').get()).toEqual({ n: 0 }) } finally { store.close() + await index.close() } }) it('finishes an interrupted deletion after reopening even when history becomes unlimited', async () => { - const root = await mkdtemp(join(tmpdir(), 'ss-retention-resume-')) - const path = join(root, 'index.sqlite') - let store = new SessionSearchStore(path) + const index = await openSessionSearchIndexFile('ss-retention-resume') + let store = new SessionSearchStore(index.path) let closed = false try { - seed(store, 1, 513, 1) + seed(index.db, 1, 513, 1) await deleteExpiredSearchFiles( - store.db, + index.db, 100, () => closed, () => {}, @@ -93,30 +91,31 @@ it('finishes an interrupted deletion after reopening even when history becomes u closed = true } ) - store = new SessionSearchStore(path) + store = new SessionSearchStore(index.path) closed = false expect(store.search({ query: 'retentionneedle' }).hits).toEqual([]) expect(store.coverage().sessionsIndexed).toBe(0) await store.purgeOlderThan(null) - expect(store.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 0 }) - expect(store.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ n: 0 }) + expect(index.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 0 }) + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ n: 0 }) } finally { if (!closed) { store.close() } - await removeTree(root) + await index.close() } }) it('does not orphan a replacement file when resuming an older deletion for the same path', async () => { - const store = new SessionSearchStore(':memory:') + const index = await openSessionSearchIndexFile('ss-retention-reused-path') + const store = new SessionSearchStore(index.path) try { - seed(store, 1, 2, 1) - store.db.exec( + seed(index.db, 1, 2, 1) + index.db.exec( "INSERT INTO search_pending_deletes(path,session_row_id) VALUES ('1',1); DELETE FROM files WHERE path='1'" ) - seed(store, 2, 2, 1) - store.db.exec("UPDATE files SET path='1' WHERE path='2'") + seed(index.db, 2, 2, 1) + index.db.exec("UPDATE files SET path='1' WHERE path='2'") await store.purgeOlderThan(100) for (const table of [ 'messages', @@ -126,28 +125,31 @@ it('does not orphan a replacement file when resuming an older deletion for the s 'files', 'search_pending_deletes' ]) { - expect(store.db.prepare(`SELECT count(*) AS n FROM ${table}`).get()).toEqual({ n: 0 }) + expect(index.db.prepare(`SELECT count(*) AS n FROM ${table}`).get()).toEqual({ n: 0 }) } } finally { store.close() + await index.close() } }) it('cancels retention between batches and resumes without exposing a partial session', async () => { - const store = new SessionSearchStore(':memory:') + const index = await openSessionSearchIndexFile('ss-retention-cancel') + const store = new SessionSearchStore(index.path) try { - seed(store, 1, 1025, 1) + seed(index.db, 1, 1025, 1) const controller = new AbortController() const purge = store.purgeOlderThan(100, controller.signal) setImmediate(() => controller.abort()) await purge - const remaining = store.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number } + const remaining = index.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number } expect(remaining.n).toBeGreaterThan(0) expect(remaining.n).toBeLessThan(1025) expect(store.search({ query: 'retentionneedle' }).hits).toEqual([]) await store.purgeOlderThan(null) - expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 0 }) + expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 0 }) } finally { store.close() + await index.close() } }) diff --git a/src/main/ai-vault-search/session-search-retention-delete.ts b/src/main/ai-vault-search/session-search-retention-delete.ts index b694e3ada02..2695e916f3b 100644 --- a/src/main/ai-vault-search/session-search-retention-delete.ts +++ b/src/main/ai-vault-search/session-search-retention-delete.ts @@ -1,4 +1,3 @@ -import { assertSearchWalBudget } from './session-search-wal-budget' import { setImmediate as yieldToEventLoop } from 'node:timers/promises' import type SyncDatabase from '../sqlite/sync-database' import { inSessionParseFileLane } from '../ai-vault/session-parse-file-lane' @@ -55,7 +54,6 @@ export async function deleteExpiredSearchFiles( if (!pending) { return } - assertSearchWalBudget(db) db.exec('BEGIN IMMEDIATE') try { const ids = db diff --git a/src/main/ai-vault-search/session-search-row-filter.ts b/src/main/ai-vault-search/session-search-row-filter.ts index 26dab02f0fb..5e351133e23 100644 --- a/src/main/ai-vault-search/session-search-row-filter.ts +++ b/src/main/ai-vault-search/session-search-row-filter.ts @@ -1,12 +1,15 @@ import { sessionSearchPathKey } from './session-search-path-key' import type { AiVaultSearchArgs } from '../../shared/ai-vault-search-types' import type { AiVaultSearchQuerySplit } from '../../shared/ai-vault-search-query-operators' -import { isRuntimePathAbsolute } from '../../shared/cross-platform-path' +import { + isRuntimePathAbsolute, + normalizeRuntimePathSeparators +} from '../../shared/cross-platform-path' /** SQL fragments for the `sessions` WHERE clause; every condition is ANDed. */ export type SessionRowFilter = { conditions: string[] - values: string[] + values: (string | number)[] } // Stored identity preserves execution-host case and WSL distro semantics. @@ -15,16 +18,28 @@ const CWD = 'cwd_key' // last segment off, leaving the parent prefix to delete out of p. const CWD_BASENAME = `replace(${CWD}, rtrim(${CWD}, replace(${CWD}, '/', '')), '')` +/** + * Every caller-supplied narrowing in one place, so `match`, `recent`, and + * `loadSessions` cannot drift apart. Row visibility is not here: it belongs to + * the `visible_sessions` / `visible_messages` views these conditions run over. + * + * Case rule, one for the whole file: a comparison that claims *identity* + * (`scopePaths`, an absolute `path:`) compares the stored key as-is, so it folds + * exactly where the execution host folds — Windows drives and the WSL distro + * segment, never a POSIX directory name. A comparison that is only a *substring + * probe* (a relative `path:`, any `repo:`) uses LIKE, which folds ASCII and + * nothing else; SQLite has no Unicode fold, and `lower()` would fold ASCII twice + * while still missing `É`, so it is not used. + */ export function sessionRowFilter( args: AiVaultSearchArgs, - split: AiVaultSearchQuerySplit + split: AiVaultSearchQuerySplit, + cutoffMs: number | null = null ): SessionRowFilter { - const filter: SessionRowFilter = { - conditions: [ - 'index_ready = 1', - 'id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL)' - ], - values: [] + const filter: SessionRowFilter = { conditions: [], values: [] } + if (cutoffMs !== null) { + filter.conditions.push('id IN (SELECT session_row_id FROM files WHERE mtime_ms >= ?)') + filter.values.push(cutoffMs) } if (args.agents && args.agents.length > 0) { filter.conditions.push(`agent IN (${args.agents.map(() => '?').join(',')})`) @@ -35,53 +50,65 @@ export function sessionRowFilter( filter.values.push(args.since) } if (args.scopePaths && args.scopePaths.length > 0) { - filter.conditions.push( - `(${args.scopePaths.map(() => `(${CWD} = ? OR substr(${CWD}, 1, length(?)) = ?)`).join(' OR ')})` + addGroup( + filter, + args.scopePaths.map((scope) => insideCondition(filter, sessionSearchPathKey(scope))) ) - for (const scope of args.scopePaths) { - // Literal prefixes keep `%` and `_` in folder names from widening scope. - const normalized = sessionSearchPathKey(scope) - filter.values.push(normalized, `${normalized}/`, `${normalized}/`) - } } - // Operators narrow, never widen: each one is its own ANDed condition on top - // of whatever scope the caller already asked for. - for (const term of split.pathTerms) { - addPathTerm(filter, term) - } - for (const term of split.repoTerms) { + // Operators narrow the caller's scope, never widen it, and follow the usual + // qualifier semantics: OR within one key, AND across keys, so `path:a path:b` + // means either while `repo:x path:a` means both. + addGroup( + filter, + split.pathTerms.map((term) => pathTermCondition(filter, term)) + ) + addGroup( + filter, // Why: a folder workspace has no repo name beyond its own folder, so the // last segment of cwd is the only honest local proxy for `repo:`. - filter.conditions.push(`lower(${CWD_BASENAME}) LIKE ? ESCAPE '\\'`) - filter.values.push(likeContains(term)) - } + split.repoTerms.map((term) => containsCondition(filter, CWD_BASENAME, term)) + ) return filter } -function addPathTerm(filter: SessionRowFilter, term: string): void { - const normalized = normalizeCwdTerm(term) - if (!normalized) { - return +function addGroup(filter: SessionRowFilter, conditions: (string | null)[]): void { + const present = conditions.filter((condition) => condition !== null) + if (present.length > 0) { + filter.conditions.push(`(${present.join(' OR ')})`) } +} + +function pathTermCondition(filter: SessionRowFilter, term: string): string | null { if (isRuntimePathAbsolute(term)) { - const key = sessionSearchPathKey(term) - filter.conditions.push(`(${CWD} = ? OR substr(${CWD}, 1, length(?)) = ?)`) - filter.values.push(key, `${key}/`, `${key}/`) - return + return insideCondition(filter, sessionSearchPathKey(term)) } - filter.conditions.push(`lower(${CWD}) LIKE ? ESCAPE '\\'`) - filter.values.push(likeContains(normalized)) + // A bare fragment cannot prove Windows semantics, so fold separators anyway: + // `path:Work\App` is a Windows user typing, never a POSIX file named `Work\App`. + const fragment = normalizeRuntimePathSeparators(term.normalize('NFC')).replace(/\/+$/, '') + return fragment ? containsCondition(filter, CWD, fragment) : null } -function normalizeCwdTerm(term: string): string { - return term.replaceAll('\\', '/').replace(/\/+$/, '').toLowerCase() +/** + * `key` itself, or anything below it. Why a half-open range and not + * `substr(key, 1, length(?)) = ?`: only `>=`/`<` can seek `sessions_cwd_key`; + * the substr form scans it. The bound is the child prefix with its last byte + * incremented, so it stops at the end of that prefix and nowhere else. The two + * arms cannot merge: one range over the bare key would also swallow a sibling + * like `/work/app-other`. No wildcards, so `%`/`_` in a folder name are literal. + */ +function insideCondition(filter: SessionRowFilter, key: string): string { + const children = `${key}/` + filter.values.push(key, children, nextAfterPrefix(children)) + return `(${CWD} = ? OR (${CWD} >= ? AND ${CWD} < ?))` } -function likeContains(value: string): string { - return `%${escapeLike(value)}%` +/** The first string that sorts after every string starting with `prefix`. */ +function nextAfterPrefix(prefix: string): string { + return prefix.slice(0, -1) + String.fromCharCode(prefix.charCodeAt(prefix.length - 1) + 1) } -// LIKE wildcards inside a user-typed term are literal text, not a pattern. -function escapeLike(value: string): string { - return value.replaceAll(/[\\%_]/g, '\\$&') +function containsCondition(filter: SessionRowFilter, column: string, term: string): string { + // LIKE wildcards inside a user-typed term are literal text, not a pattern. + filter.values.push(`%${term.normalize('NFC').replaceAll(/[\\%_]/g, '\\$&')}%`) + return `${column} LIKE ? ESCAPE '\\'` } diff --git a/src/main/ai-vault-search/session-search-schema.test.ts b/src/main/ai-vault-search/session-search-schema.test.ts index 4e62ed8c7f8..c74a9cd853d 100644 --- a/src/main/ai-vault-search/session-search-schema.test.ts +++ b/src/main/ai-vault-search/session-search-schema.test.ts @@ -1,15 +1,34 @@ +import type * as NodeFs from 'node:fs' import { mkdtemp, stat, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' -import { removeTree } from '../../shared/windows-transient-lock-removal' +import { + removeTree, + WINDOWS_RM_MAX_RETRIES, + WINDOWS_RM_RETRY_DELAY_MS +} from '../../shared/windows-transient-lock-removal' import SyncDatabase from '../sqlite/sync-database' import { SESSION_SEARCH_SCHEMA_VERSION, openSessionSearchDatabase, - removeSessionSearchDatabase + removeSessionSearchDatabase, + VISIBLE_MESSAGES, + VISIBLE_SESSIONS } from './session-search-schema' +const recordedRmSync = vi.hoisted(() => vi.fn()) +vi.mock('node:fs', async () => { + const actual = await vi.importActual('node:fs') + return { + ...actual, + rmSync: (...args: Parameters) => { + recordedRmSync(...args) + return actual.rmSync(...args) + } + } +}) + let roots: string[] = [] afterEach(async () => { @@ -99,3 +118,48 @@ it('closes the SQLite handle when corrupt data fails initialization', async () = const recovered = openSessionSearchDatabase(path) recovered.close() }) + +describe('visibility views', () => { + it('hides a staging session, a tombstoned session and an in-flight batch', async () => { + const db = openSessionSearchDatabase(await tempDatabasePath()) + try { + db.exec(`INSERT INTO sessions(id,index_ready,agent,session_id,file_path,title,resume_command) + VALUES (1,1,'claude','a','a','published',''),(2,0,'claude','b','b','staging',''), + (3,1,'claude','c','c','tombstoned',''); + INSERT INTO search_pending_deletes(path,session_row_id) VALUES ('c',3); + INSERT INTO search_write_batches(id,session_row_id) VALUES (7,1); + INSERT INTO messages(id,session_row_id,batch_id,role) VALUES (1,1,NULL,'user'),(2,1,7,'user')`) + expect(db.prepare(`SELECT title FROM ${VISIBLE_SESSIONS} ORDER BY id`).all()).toEqual([ + { title: 'published' } + ]) + expect(db.prepare(`SELECT id FROM ${VISIBLE_MESSAGES} ORDER BY id`).all()).toEqual([ + { id: 1 } + ]) + // Publish clears the pointer, so visibility never depends on the batch row surviving. + db.exec( + 'UPDATE messages SET batch_id=NULL WHERE batch_id=7; DELETE FROM search_write_batches' + ) + expect(db.prepare(`SELECT count(*) AS n FROM ${VISIBLE_MESSAGES}`).get()).toEqual({ n: 2 }) + } finally { + db.close() + } + }) +}) + +it('gives Windows the shared retry options for a late handle release', async () => { + const path = await tempDatabasePath() + vi.spyOn(process, 'platform', 'get').mockReturnValue('win32') + recordedRmSync.mockClear() + try { + removeSessionSearchDatabase(path) + expect(recordedRmSync).toHaveBeenCalled() + for (const [, options] of recordedRmSync.mock.calls) { + expect(options).toMatchObject({ + maxRetries: WINDOWS_RM_MAX_RETRIES, + retryDelay: WINDOWS_RM_RETRY_DELAY_MS + }) + } + } finally { + vi.restoreAllMocks() + } +}) diff --git a/src/main/ai-vault-search/session-search-schema.ts b/src/main/ai-vault-search/session-search-schema.ts index 82302bb39ce..bdc7ccb2828 100644 --- a/src/main/ai-vault-search/session-search-schema.ts +++ b/src/main/ai-vault-search/session-search-schema.ts @@ -1,8 +1,9 @@ import { rmSync } from 'node:fs' import SyncDatabase from '../sqlite/sync-database' +import { transientLockRemovalOptions } from '../../shared/windows-transient-lock-removal' // Bump to drop and rebuild: the index is a cache over the transcripts, never a source. -export const SESSION_SEARCH_SCHEMA_VERSION = 9 +export const SESSION_SEARCH_SCHEMA_VERSION = 10 // unicode61 keeps `_ . - /` inside tokens so paths and identifiers match exactly; // the `identifiers` column carries the split form (see session-search-identifier-split). @@ -10,6 +11,10 @@ export const SESSION_SEARCH_SCHEMA_VERSION = 9 // left out so `#123` still answers a search for `123`. const TOKENIZER = `tokenize="unicode61 tokenchars '_.-/+'"` +/** Sessions and messages a read may return: published, not tombstoned. */ +export const VISIBLE_SESSIONS = 'visible_sessions' +export const VISIBLE_MESSAGES = 'visible_messages' + const SCHEMA_SQL = ` CREATE TABLE IF NOT EXISTS meta(key TEXT PRIMARY KEY, value TEXT NOT NULL); CREATE TABLE IF NOT EXISTS sessions( @@ -35,6 +40,7 @@ CREATE TABLE IF NOT EXISTS sessions( CREATE INDEX IF NOT EXISTS sessions_agent ON sessions(agent); CREATE INDEX IF NOT EXISTS sessions_content_hash ON sessions(content_hash); CREATE INDEX IF NOT EXISTS sessions_updated_at ON sessions(updated_at); +CREATE INDEX IF NOT EXISTS sessions_cwd_key ON sessions(cwd_key); CREATE TABLE IF NOT EXISTS files( path TEXT PRIMARY KEY, dev INTEGER, @@ -49,13 +55,12 @@ CREATE TABLE IF NOT EXISTS search_pending_deletes( session_row_id INTEGER NOT NULL, batch_id INTEGER ); +-- A row exists only while its batch is in flight; publish clears its messages and deletes it. CREATE TABLE IF NOT EXISTS search_write_batches( id INTEGER PRIMARY KEY, - session_row_id INTEGER NOT NULL, - published INTEGER NOT NULL DEFAULT 0 + session_row_id INTEGER NOT NULL ); CREATE INDEX IF NOT EXISTS search_write_batches_session ON search_write_batches(session_row_id); -CREATE INDEX IF NOT EXISTS search_write_batches_pending ON search_write_batches(published); CREATE TABLE IF NOT EXISTS messages( id INTEGER PRIMARY KEY, session_row_id INTEGER NOT NULL, @@ -80,6 +85,13 @@ CREATE TABLE IF NOT EXISTS search_log( hits INTEGER NOT NULL, duration_ms REAL NOT NULL ); +-- Why: staged rows must never reach a result. One definition per half, so a new +-- read site cannot forget one; SQLite flattens both into the caller's plan. +CREATE VIEW IF NOT EXISTS ${VISIBLE_SESSIONS} AS SELECT * FROM sessions + WHERE index_ready = 1 + AND id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL); +CREATE VIEW IF NOT EXISTS ${VISIBLE_MESSAGES} AS SELECT * FROM messages + WHERE batch_id IS NULL; ` export function openSessionSearchDatabase(path: string): SyncDatabase { @@ -122,16 +134,10 @@ export function removeSessionSearchDatabase(path: string): void { return } for (const suffix of ['', '-wal', '-shm', '-journal']) { - rmSync(`${path}${suffix}`, { force: true }) + rmSync(`${path}${suffix}`, transientLockRemovalOptions()) } } -export function openSessionSearchDatabaseReadOnly(path: string): SyncDatabase { - const db = new SyncDatabase(path, { readonly: true, fileMustExist: true }) - db.pragma('busy_timeout = 1500') - return db -} - function readSchemaVersion(db: SyncDatabase): number | null { const table = db .prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'meta'") diff --git a/src/main/ai-vault-search/session-search-service.ts b/src/main/ai-vault-search/session-search-service.ts index 94a5680eed4..d8e6a8af977 100644 --- a/src/main/ai-vault-search/session-search-service.ts +++ b/src/main/ai-vault-search/session-search-service.ts @@ -8,7 +8,10 @@ import type { AiVaultSearchCoverage, AiVaultSearchResult } from '../../shared/ai-vault-search-types' -import { DISABLED_AI_VAULT_SEARCH_COVERAGE as DISABLED_COVERAGE } from '../../shared/ai-vault-search-coverage' +import { + DISABLED_AI_VAULT_SEARCH_COVERAGE as DISABLED_COVERAGE, + NO_AI_VAULT_SEARCH_INDEX_RESULT as NO_INDEX_RESULT +} from '../../shared/ai-vault-search-coverage' import { aiVaultSearchHistoryCutoffMs, narrowsAiVaultSearchHistory, @@ -20,7 +23,10 @@ import { sessionCandidatesFromDiscoveries } from '../ai-vault/session-scanner-ca import { discoverAiVaultSessionSources } from '../ai-vault/session-scanner-source-discovery' import type { AiVaultScanIssue } from '../../shared/ai-vault-types' import type { AiVaultScanOptions, SessionFileCandidate } from '../ai-vault/session-scanner-types' -import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture' +import { + getSessionSearchIndexSink, + registerSessionSearchIndexSink +} from '../ai-vault/session-search-capture' import { parseSearchCandidates } from './session-search-parse-candidates' import { removeSessionSearchDatabase } from './session-search-schema' import { SessionSearchStore } from './session-search-store' @@ -53,7 +59,7 @@ export class SessionSearchService { this.databasePath = options.databasePath this.policy = { ...options } if (this.policy.enabled) { - this.open() + this.applyPolicyToStore(this.openStore()) } } @@ -63,9 +69,8 @@ export class SessionSearchService { roots: SessionSearchScanRoots, signal?: AbortSignal ): Promise { - const store = this.store - if (!store) { - return { hits: [], route: 'and', durationMs: 0, coverage: DISABLED_COVERAGE } + if (!this.store) { + return NO_INDEX_RESULT } const backfill = this.ensureBackfill(roots) // Why: the backfill parses in this same process and an 80 MB transcript @@ -80,7 +85,7 @@ export class SessionSearchService { async (sharedSignal) => { await this.parseAll( this.withinHistory(await discoverRecentSearchFiles(roots, sharedSignal)), - sharedSignal + { signal: sharedSignal } ) await this.reindexStale(sharedSignal) }, @@ -88,9 +93,12 @@ export class SessionSearchService { ) } void backfill + // Why: presence checks await between query rounds, and a configure() in + // that window closes the database; read the handle per round so a closed + // store yields no hits instead of throwing on a freed statement. return await searchPresentSessionSources( args, - (query) => store.search(query), + (query) => this.store?.search(query) ?? NO_INDEX_RESULT, (paths) => this.invalidate(paths), signal ) @@ -138,8 +146,6 @@ export class SessionSearchService { const wasEnabled = this.policy.enabled const previousDays = this.policy.historyDays this.policy = { ...next } - this.store?.indexing.setPaused(next.paused === true) - this.store?.setHistoryDays(next.historyDays) if (options.clearIndex || (wasEnabled && !next.enabled)) { await this.stop() } else if (wasEnabled && (next.paused || previousDays !== next.historyDays)) { @@ -152,9 +158,8 @@ export class SessionSearchService { if (!next.enabled) { return DISABLED_COVERAGE } - const store = this.store ?? this.open() - store.setAcceptingWrites(!next.paused) - store.indexing.setPaused(next.paused === true) + const store = this.store ?? this.openStore() + this.applyPolicyToStore(store) const cutoff = aiVaultSearchHistoryCutoffMs(next.historyDays) if ( wasEnabled && @@ -208,23 +213,24 @@ export class SessionSearchService { } } - dispose(): void { - this.backfillController?.abort() - this.closeStore() - } - async close(): Promise { await this.stop({ drainRefreshes: true }) } - private open(): SessionSearchStore { + /** Creation only; the caller applies the policy before anything can await. */ + private openStore(): SessionSearchStore { mkdirSync(dirname(this.databasePath), { recursive: true }) - this.store = new SessionSearchStore(this.databasePath) - this.store.setHistoryDays(this.policy.historyDays) - this.store.setAcceptingWrites(!this.policy.paused) - this.store.indexing.setPaused(this.policy.paused === true) - registerSessionSearchIndexSink(this.store) - return this.store + const store = new SessionSearchStore(this.databasePath) + this.store = store + registerSessionSearchIndexSink(store) + return store + } + + /** The only writer of policy-derived store state, so the bits cannot drift. */ + private applyPolicyToStore(store: SessionSearchStore): void { + store.setHistoryDays(this.policy.historyDays) + store.setAcceptingWrites(!this.policy.paused) + store.indexing.setPaused(this.policy.paused === true) } /** Waits for the aborted backfill so its last parse cannot write to a closed store. */ @@ -247,9 +253,6 @@ export class SessionSearchService { } } finally { this.stopping = false - if (options.keepStore) { - this.store?.setAcceptingWrites(!this.policy.paused) - } } } @@ -260,7 +263,11 @@ export class SessionSearchService { if (!store) { return } - registerSessionSearchIndexSink(null) + // Why: shutdown is async, so a replacement service may already own the sink; + // clearing it unconditionally would silently stop feeding the new index. + if (getSessionSearchIndexSink() === store) { + registerSessionSearchIndexSink(null) + } store.close() } @@ -271,7 +278,7 @@ export class SessionSearchService { return } try { - await this.parseAll(this.withinHistory(stale), signal) + await this.parseAll(this.withinHistory(stale), { signal }) } catch (error) { // Why: a cancelled search (the renderer retires them per keystroke) must // not lose the queue; whatever did not get parsed goes back for next time. @@ -312,7 +319,7 @@ export class SessionSearchService { recordSearchDiscovered(store, discoveries, issues) const eligible = this.withinHistory(candidates) store.indexing.discovered(eligible.length, issues.length) - await this.parseAll(eligible, signal, { yieldToSearches: signal }) + await this.parseAll(eligible, { signal, backfillSignal: signal }) store.indexing.finish() store.setBackfillState('complete') } catch (error) { @@ -322,20 +329,26 @@ export class SessionSearchService { } } + /** `backfillSignal` marks the long tail: those files report progress and yield to searches. */ private async parseAll( candidates: SessionFileCandidate[], - signal?: AbortSignal, - options: { yieldToSearches?: AbortSignal } = {} + options: { signal?: AbortSignal; backfillSignal?: AbortSignal } ): Promise { - if (this.store) { - await parseSearchCandidates( - this.store, - candidates, - signal, - options.yieldToSearches - ? () => this.waitForIdleSearches(options.yieldToSearches!) - : undefined - ) + const store = this.store + const backfillSignal = options.backfillSignal + if (!store) { + return } + await parseSearchCandidates(store, candidates, { + signal: options.signal, + ...(backfillSignal + ? { + onFileProcessed: async (failed: boolean) => { + store.indexing.processed(failed) + await this.waitForIdleSearches(backfillSignal) + } + } + : {}) + }) } } diff --git a/src/main/ai-vault-search/session-search-session-row.ts b/src/main/ai-vault-search/session-search-session-row.ts deleted file mode 100644 index 748ecc2fd83..00000000000 --- a/src/main/ai-vault-search/session-search-session-row.ts +++ /dev/null @@ -1,33 +0,0 @@ -import type { AiVaultAgent } from '../../shared/ai-vault-types' -import type { AiVaultSearchHit } from '../../shared/ai-vault-search-types' - -export type SessionRow = { - id: number - agent: AiVaultAgent - session_id: string - file_path: string - codex_home: string | null - title: string - cwd: string | null - branch: string | null - updated_at: string | null - message_count: number - resume_command: string - content_hash: string | null - content_hash_count: number -} - -export function sessionFields(session: SessionRow): Omit { - return { - agent: session.agent, - sessionId: session.session_id, - filePath: session.file_path, - codexHome: session.codex_home, - title: session.title, - cwd: session.cwd, - branch: session.branch, - updatedAt: session.updated_at, - messageCount: session.message_count, - resumeCommand: session.resume_command - } -} diff --git a/src/main/ai-vault-search/session-search-source-presence.test.ts b/src/main/ai-vault-search/session-search-source-presence.test.ts index 7574cc209f2..23f5a61b16a 100644 --- a/src/main/ai-vault-search/session-search-source-presence.test.ts +++ b/src/main/ai-vault-search/session-search-source-presence.test.ts @@ -58,7 +58,7 @@ it('checks individual OpenCode identities when several sessions share one databa } }) -it('omits unreadable sources without treating them as confirmed deletion', async () => { +it('keeps unreadable sources as hits and flags them instead of dropping them', async () => { const root = await mkdtemp(join(tmpdir(), 'orca-search-unreadable-')) try { const filePath = join(root, 'opencode.db') @@ -73,7 +73,8 @@ it('omits unreadable sources without treating them as confirmed deletion', async } as AiVaultSearchResult, invalidate ) - expect(result).toMatchObject({ hits: [], sourceUnavailableFiles: 1 }) + expect(result.hits.map((hit) => hit.sessionId)).toEqual(['fixture']) + expect(result).toMatchObject({ sourceUnavailableFiles: 1 }) expect(invalidate).not.toHaveBeenCalled() } finally { await rm(root, { recursive: true, force: true }) diff --git a/src/main/ai-vault-search/session-search-source-presence.ts b/src/main/ai-vault-search/session-search-source-presence.ts index db1862c6733..893b3e07ccc 100644 --- a/src/main/ai-vault-search/session-search-source-presence.ts +++ b/src/main/ai-vault-search/session-search-source-presence.ts @@ -100,7 +100,7 @@ async function checkSearchSources( throwIfSignalAborted(signal) } -/** Refill limited results after filtering, without deleting unreadable sources or unbounded scans. */ +/** Refill limited results after dropping deleted sources, without unbounded scans. */ export async function searchPresentSessionSources( args: AiVaultSearchArgs, search: (args: AiVaultSearchArgs) => AiVaultSearchResult, @@ -120,7 +120,10 @@ export async function searchPresentSessionSources( const result = search({ ...args, limit }) durationMs += result.durationMs await checkSearchSources(result.hits, known, unavailable, invalidate, signal) - const hits = result.hits.filter((hit) => known.get(candidatePath(hit)) === 'present') + // Why: loss of contact is not evidence of absence (see + // docs/reference/ssh-execution-boundary.md). Only a proven deletion drops a + // hit; an unreadable source is still shown, counted in sourceUnavailableFiles. + const hits = result.hits.filter((hit) => known.get(candidatePath(hit)) !== 'missing') const missing = result.hits.some((hit) => known.get(candidatePath(hit)) === 'missing') const exhausted = result.hits.length < limit && !missing const budgetExhausted = diff --git a/src/main/ai-vault-search/session-search-source-refill.test.ts b/src/main/ai-vault-search/session-search-source-refill.test.ts index de38af535ad..d2ce14e8565 100644 --- a/src/main/ai-vault-search/session-search-source-refill.test.ts +++ b/src/main/ai-vault-search/session-search-source-refill.test.ts @@ -4,7 +4,7 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import * as transcriptFs from '../native-chat/wsl-transcript-fs-access' import { SessionSearchStore } from './session-search-store' -import { stagedWriteUpdate } from './session-search-staged-write-fixtures' +import { stagedWriteUpdate } from './session-search-staged-write-test-fixture' import { searchPresentSessionSources } from './session-search-source-presence' import type { AiVaultSearchArgs, AiVaultSearchResult } from '../../shared/ai-vault-search-types' @@ -49,7 +49,7 @@ it('refills a deleted top hit in the first query with a real index', async () => expect(result.omittedHits).toBeUndefined() }) -it('refills past an unreadable top hit without deleting it or recounting it', async () => { +it('keeps an unreadable top hit instead of refilling past it or deleting it', async () => { const { store, search, invalidate } = await fixture() const args = { query: 'refillneedle', limit: 1 } const top = store.search(args).hits[0] @@ -63,15 +63,15 @@ it('refills past an unreadable top hit without deleting it or recounting it', as return stat(path, priority, signal) }) const result = await searchPresentSessionSources(args, search, invalidate) - expect(result.hits).toHaveLength(1) - expect(result.hits[0].filePath).not.toBe(top.filePath) + // Loss of contact is not evidence of absence: the hit stays, flagged. + expect(result.hits.map((hit) => hit.filePath)).toEqual([top.filePath]) expect(result.sourceUnavailableFiles).toBe(1) expect(invalidate).not.toHaveBeenCalled() expect(store.search({ query: 'refillneedle' }).hits).toHaveLength(2) expect(probe.mock.calls.filter(([path]) => path === top.filePath)).toHaveLength(1) }) -it('bounds refill and reports omissions when unavailable hits consume the budget', async () => { +it('bounds refill and reports omissions when deleted hits consume the budget', async () => { const { store } = await fixture() const base = store.search({ query: 'refillneedle' }) const source = base.hits[0] @@ -80,7 +80,7 @@ it('bounds refill and reports omissions when unavailable hits consume the budget filePath: join(source.filePath, String(i)) })) vi.spyOn(transcriptFs, 'wslGatedStat').mockRejectedValue( - Object.assign(new Error('denied'), { code: 'EACCES' }) + Object.assign(new Error('gone'), { code: 'ENOENT' }) ) const search = vi.fn((args: AiVaultSearchArgs) => ({ ...base, hits: hits.slice(0, args.limit) })) const invalidate = vi.fn() @@ -91,8 +91,9 @@ it('bounds refill and reports omissions when unavailable hits consume the budget ) expect(search).toHaveBeenCalledTimes(4) expect(search.mock.calls.map(([args]) => args.limit)).toEqual([1, 2, 4, 8]) - expect(result).toMatchObject({ hits: [], omittedHits: 8, sourceUnavailableFiles: 8 }) - expect(invalidate).not.toHaveBeenCalled() + expect(result).toMatchObject({ hits: [], omittedHits: 8 }) + expect(result.sourceUnavailableFiles).toBeUndefined() + expect(invalidate).toHaveBeenCalled() }) it('does not invalidate a WSL source when its share reports ENOENT', async () => { @@ -106,7 +107,7 @@ it('does not invalidate a WSL source when its share reports ENOENT', async () => const invalidate = vi.fn() expect( await searchPresentSessionSources({ query: 'refillneedle' }, () => result, invalidate) - ).toMatchObject({ hits: [], sourceUnavailableFiles: 1 }) + ).toMatchObject({ hits: [{ filePath }], sourceUnavailableFiles: 1 }) expect(invalidate).not.toHaveBeenCalled() }) diff --git a/src/main/ai-vault-search/session-search-staged-write-fixtures.ts b/src/main/ai-vault-search/session-search-staged-write-fixtures.ts deleted file mode 100644 index 5e4c1ea54d1..00000000000 --- a/src/main/ai-vault-search/session-search-staged-write-fixtures.ts +++ /dev/null @@ -1,42 +0,0 @@ -import type { SessionSearchIndexUpdate } from '../ai-vault/session-search-capture' - -export function stagedWriteUpdate( - text: string, - count: number, - mode: 'append' | 'replace' = 'replace' -): SessionSearchIndexUpdate { - const at = new Date().toISOString() - return { - candidate: { - agent: 'claude', - codexHome: null, - file: { path: 'synthetic-transcript', mtimeMs: Date.now(), modifiedAt: at, sizeBytes: 2 } - }, - session: { - id: 'fixture', - executionHostId: 'local', - agent: 'claude', - sessionId: 'fixture', - title: text, - cwd: '/fixture', - branch: null, - model: null, - filePath: 'synthetic-transcript', - codexHome: null, - createdAt: at, - updatedAt: at, - modifiedAt: at, - messageCount: count, - totalTokens: 0, - previewMessages: [], - queuedMessageCount: 0, - subagentTranscriptCount: 0, - resumeCommand: '', - subagent: null - }, - mode, - messages: Array.from({ length: count }, () => ({ role: 'user', text, timestamp: null })), - previousByteOffset: mode === 'append' ? 1 : 0, - byteOffset: mode === 'append' ? 2 : 1 - } -} diff --git a/src/main/ai-vault-search/session-search-staged-write-test-fixture.ts b/src/main/ai-vault-search/session-search-staged-write-test-fixture.ts new file mode 100644 index 00000000000..84c352c480a --- /dev/null +++ b/src/main/ai-vault-search/session-search-staged-write-test-fixture.ts @@ -0,0 +1,87 @@ +import { mkdtemp } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { removeTree } from '../../shared/windows-transient-lock-removal' +import type { + SessionSearchIndexResult, + SessionSearchIndexUpdate +} from '../ai-vault/session-search-capture' +import type { AiVaultSession } from '../../shared/ai-vault-types' +import type SyncDatabase from '../sqlite/sync-database' +import { openSessionSearchDatabase } from './session-search-schema' + +/** Carries `session`/`byteOffset` alongside the write so assertions can read the expected result. */ +export function stagedWriteUpdate( + text: string, + count: number, + mode: 'append' | 'replace' = 'replace' +): SessionSearchIndexUpdate & { result: Promise } { + const at = new Date().toISOString() + const session: AiVaultSession = { + id: 'fixture', + executionHostId: 'local', + agent: 'claude', + sessionId: 'fixture', + title: text, + cwd: '/fixture', + branch: null, + model: null, + filePath: 'synthetic-transcript', + codexHome: null, + createdAt: at, + updatedAt: at, + modifiedAt: at, + messageCount: count, + totalTokens: 0, + previewMessages: [], + queuedMessageCount: 0, + subagentTranscriptCount: 0, + resumeCommand: '', + subagent: null + } + const update: SessionSearchIndexUpdate & { result: Promise } = { + candidate: { + agent: 'claude', + codexHome: null, + file: { path: 'synthetic-transcript', mtimeMs: Date.now(), modifiedAt: at, sizeBytes: 2 } + }, + session, + mode, + messages: Array.from({ length: count }, () => ({ role: 'user', text, timestamp: null })), + previousByteOffset: mode === 'append' ? 1 : 0, + byteOffset: mode === 'append' ? 2 : 1, + // Why: callers retarget `session`/`byteOffset` after construction, so the + // promise the writer awaits must read them then, not at build time. + result: Promise.resolve().then(() => ({ + session: update.session, + byteOffset: update.byteOffset + })) + } + return update +} + +export type SessionSearchIndexFile = { + path: string + /** The store keeps its own connection private, so row assertions need this one. */ + db: SyncDatabase + close: () => Promise +} + +/** An on-disk index: `:memory:` is per-connection, so a second reader needs a real file. */ +export async function openSessionSearchIndexFile(name: string): Promise { + const root = await mkdtemp(join(tmpdir(), `${name}-`)) + const path = join(root, 'index.sqlite') + const db = openSessionSearchDatabase(path) + let open = true + return { + path, + db, + close: async () => { + if (open) { + open = false + db.close() + } + await removeTree(root) + } + } +} diff --git a/src/main/ai-vault-search/session-search-staged-write.test.ts b/src/main/ai-vault-search/session-search-staged-write.test.ts index 40b6691d358..d4b870048aa 100644 --- a/src/main/ai-vault-search/session-search-staged-write.test.ts +++ b/src/main/ai-vault-search/session-search-staged-write.test.ts @@ -1,32 +1,38 @@ -import { removeTree } from '../../shared/windows-transient-lock-removal' -import { mkdtemp } from 'node:fs/promises' -import { join } from 'node:path' -import { tmpdir } from 'node:os' import { expect, it } from 'vitest' import { SessionSearchStore } from './session-search-store' import { SessionSearchIndexWriter, SEARCH_WRITE_ROWS_PER_STEP } from './session-search-index-writer' -import { stagedWriteUpdate as update } from './session-search-staged-write-fixtures' +import { + openSessionSearchIndexFile, + stagedWriteUpdate as update +} from './session-search-staged-write-test-fixture' + +function stagedRows(db: { prepare: (sql: string) => { get: () => unknown } }): number { + return ( + db + .prepare( + 'SELECT count(*) AS n FROM messages WHERE batch_id IN (SELECT id FROM search_write_batches)' + ) + .get() as { n: number } + ).n +} + +function messageCount(db: { prepare: (sql: string) => { get: () => unknown } }): number { + return (db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n +} it.each(['replace', 'append'] as const)( 'publishes a large %s atomically after bounded steps', async (mode) => { - const store = new SessionSearchStore(':memory:') - const writer = new SessionSearchIndexWriter(store.db) + const index = await openSessionSearchIndexFile('ss-staged-publish') + const store = new SessionSearchStore(index.path) + const writer = new SessionSearchIndexWriter(index.db) try { await writer.apply(update('oldneedle', 1)) let steps = 0, previous = 0 - const applied = await writer.apply( - update('newneedle', 1000, mode), - () => true, - async () => { - const rows = ( - store.db - .prepare( - 'SELECT count(*) AS n FROM messages WHERE batch_id IN (SELECT id FROM search_write_batches WHERE published=0)' - ) - .get() as { n: number } - ).n + const applied = await writer.apply(update('newneedle', 1000, mode), { + yieldStep: async () => { + const rows = stagedRows(index.db) expect(rows - previous).toBeLessThanOrEqual(SEARCH_WRITE_ROWS_PER_STEP) previous = rows steps++ @@ -34,18 +40,17 @@ it.each(['replace', 'append'] as const)( expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1) expect(writer.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(1) } - ) + }) expect(applied).toBe(true) expect(steps).toBe(8) expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1) expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(mode === 'append' ? 1 : 0) expect(store.search({ query: 'newneedle' }).hits[0].title).toBe('newneedle') await store.purgeOlderThan(null) - expect( - (store.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n - ).toBe(mode === 'append' ? 1001 : 1000) + expect(messageCount(index.db)).toBe(mode === 'append' ? 1001 : 1000) } finally { store.close() + await index.close() } } ) @@ -53,93 +58,120 @@ it.each(['replace', 'append'] as const)( it.each(['replace', 'append'] as const)( 'recovers an interrupted %s without publishing rows or advancing its cursor', async (mode) => { - const root = await mkdtemp(join(tmpdir(), 'ss-staged-')) - const path = join(root, 'index.sqlite') - let store = new SessionSearchStore(path), - open = true + const index = await openSessionSearchIndexFile('ss-staged-recover') + let store = new SessionSearchStore(index.path) + let open = true try { - const writer = new SessionSearchIndexWriter(store.db) + const writer = new SessionSearchIndexWriter(index.db) await writer.apply(update('oldneedle', 1)) expect( - await writer.apply( - update('newneedle', 1000, mode), - () => open, - async () => { + await writer.apply(update('newneedle', 1000, mode), { + active: () => open, + yieldStep: async () => { store.close() open = false } - ) + }) ).toBe(false) - store = new SessionSearchStore(path) + store = new SessionSearchStore(index.path) open = true expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0) expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1) expect(store.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(1) await store.purgeOlderThan(null) - expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1 }) + expect(messageCount(index.db)).toBe(1) await store.apply(update('newneedle', 1000, mode)) expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1) } finally { if (open) { store.close() } - await removeTree(root) + await index.close() } } ) it('does not resurrect a newly invalidated file or retain a cancelled append', async () => { - const store = new SessionSearchStore(':memory:') - const writer = new SessionSearchIndexWriter(store.db) + const index = await openSessionSearchIndexFile('ss-staged-invalidated') + const store = new SessionSearchStore(index.path) + const writer = new SessionSearchIndexWriter(index.db) try { expect( - await writer.apply( - update('newneedle', 1000), - () => true, - async () => { + await writer.apply(update('newneedle', 1000), { + yieldStep: async () => { writer.removeFile('synthetic-transcript') } - ) + }) ).toBe(false) expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0) await store.purgeOlderThan(null) await writer.apply(update('oldneedle', 1)) let accepted = true expect( - await writer.apply( - update('newneedle', 1000, 'append'), - () => accepted, - async () => { + await writer.apply(update('newneedle', 1000, 'append'), { + active: () => accepted, + yieldStep: async () => { accepted = false }, - () => true - ) + available: () => true + }) ).toBe(false) await store.purgeOlderThan(null) - expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1 }) + expect(messageCount(index.db)).toBe(1) expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1) } finally { store.close() + await index.close() } }) it('does not suggest unpublished vocabulary while a large update is being written', async () => { - const store = new SessionSearchStore(':memory:') - const writer = new SessionSearchIndexWriter(store.db) + const index = await openSessionSearchIndexFile('ss-staged-vocab') + const store = new SessionSearchStore(index.path) + const writer = new SessionSearchIndexWriter(index.db) try { await writer.apply(update('oldneedle', 1)) - await writer.apply( - update('coalesces', 1000, 'append'), - () => true, - async () => { + await writer.apply(update('coalesces', 1000, 'append'), { + yieldStep: async () => { const result = store.search({ query: 'coalescs' }) expect(result.hits).toHaveLength(0) expect(result.repairedTerms).toBeUndefined() } - ) + }) expect(store.search({ query: 'coalescs' }).repairedTerms).toEqual(['coalesces']) } finally { store.close() + await index.close() + } +}) + +it('keeps published rows visible when a later batch reuses the freed id', async () => { + const index = await openSessionSearchIndexFile('ss-staged-batch-reuse') + const store = new SessionSearchStore(index.path) + const writer = new SessionSearchIndexWriter(index.db) + const inFlightBatchId = (): number => + (index.db.prepare('SELECT id FROM search_write_batches').get() as { id: number }).id + try { + let published: number | null = null + await writer.apply(update('oldneedle', 1000), { + yieldStep: async () => { + published ??= inFlightBatchId() + } + }) + let reused: number | null = null + await writer.apply(update('newneedle', 1000, 'append'), { + yieldStep: async () => { + reused ??= inFlightBatchId() + expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1) + } + }) + // Without this the test proves nothing: SQLite hands the freed rowid straight back. + expect(reused).toBe(published) + expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1) + expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1) + } finally { + store.close() + await index.close() } }) @@ -153,7 +185,7 @@ it('keeps published data and queues recovery when an append cursor is stale', as expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1) expect(store.search({ query: 'staleneedle' }).hits).toHaveLength(0) expect(store.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(2) - expect(store.staleCount).toBe(1) + expect(store.coverage().filesPending).toBe(1) } finally { store.close() } diff --git a/src/main/ai-vault-search/session-search-store.test.ts b/src/main/ai-vault-search/session-search-store.test.ts index c1103135590..5a6a60658d7 100644 --- a/src/main/ai-vault-search/session-search-store.test.ts +++ b/src/main/ai-vault-search/session-search-store.test.ts @@ -7,6 +7,7 @@ import { registerSessionSearchIndexSink, withSessionSearchIndexRequired } from '../ai-vault/session-search-capture' +import SyncDatabase from '../sqlite/sync-database' import { SessionSearchStore } from './session-search-store' import { assistantRecord, @@ -16,11 +17,13 @@ import { } from './session-search-transcript-fixtures' let tempRoots: string[] = [] let store: SessionSearchStore +let databasePath: string beforeEach(async () => { resetSessionParseCacheForTests() const root = await makeTempDir() - store = new SessionSearchStore(join(root, 'index.sqlite'), (error) => { + databasePath = join(root, 'index.sqlite') + store = new SessionSearchStore(databasePath, (error) => { throw error }) registerSessionSearchIndexSink(store) @@ -279,14 +282,19 @@ describe('SessionSearchStore', () => { await writeFile(path, `${userRecord(0, `padding ${filler}`, id)}\n`) await parse(path) } - const pageCount = (): number => Number(store.db.pragma('page_count', { simple: true })) - const before = pageCount() + const reader = new SyncDatabase(databasePath, { readonly: true }) + try { + const pageCount = (): number => Number(reader.pragma('page_count', { simple: true })) + const before = pageCount() - await store.purgeOlderThan(Date.now() + 60_000) + await store.purgeOlderThan(Date.now() + 60_000) - expect(store.coverage().sessionsIndexed).toBe(0) - expect(Number(store.db.pragma('freelist_count', { simple: true }))).toBe(0) - expect(pageCount()).toBeLessThan(before) + expect(store.coverage().sessionsIndexed).toBe(0) + expect(Number(reader.pragma('freelist_count', { simple: true }))).toBe(0) + expect(pageCount()).toBeLessThan(before) + } finally { + reader.close() + } }) it('warms once and survives a close mid-way', async () => { diff --git a/src/main/ai-vault-search/session-search-store.ts b/src/main/ai-vault-search/session-search-store.ts index a89eb205d9f..dc2acd21a48 100644 --- a/src/main/ai-vault-search/session-search-store.ts +++ b/src/main/ai-vault-search/session-search-store.ts @@ -1,6 +1,8 @@ import { SessionSearchIndexingProgress } from './session-search-indexing-progress' -import { SessionSearchMaintenance } from './session-search-maintenance' -import { recoverSearchWrites } from './session-search-write-recovery' +import { compactSessionSearchIndex } from './session-search-index-compaction' +import { logSessionSearchQuery } from './session-search-query-log' +import { warmSessionSearchPages } from './session-search-page-warmup' +import { recoverSearchWrites } from './session-search-pending-deletes' import { deleteExpiredSearchFiles } from './session-search-retention-delete' import type { AiVaultAgent } from '../../shared/ai-vault-types' import { aiVaultSearchHistoryCutoffMs } from '../../shared/ai-vault-search-settings' @@ -20,19 +22,26 @@ import type { import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' import { SessionSearchIndexWriter, type SessionSearchMetadata } from './session-search-index-writer' import { SessionSearchQuery } from './session-search-query' -import { openSessionSearchDatabase } from './session-search-schema' +import { + openSessionSearchDatabase, + VISIBLE_MESSAGES, + VISIBLE_SESSIONS +} from './session-search-schema' export type SessionSearchBackfillState = 'idle' | 'running' | 'complete' type ProviderDiscovery = { files: number; parseFailures: number; scanIssues: number } +export type SessionSearchStoreOptions = { + /** The WAL backlog a staging write refuses to grow past. Only tests narrow it. */ + walBudgetBytes?: number +} + /** Owns the index database: the scanner writes through it, search reads from it. */ export class SessionSearchStore implements SessionSearchIndexSink { - readonly streamingCapture = true readonly indexing = new SessionSearchIndexingProgress() private writeEpoch = 0 - /** Exposed for tests that assert on file-level state (page counts). */ - readonly db: SyncDatabase + private readonly db: SyncDatabase private readonly writer: SessionSearchIndexWriter private readonly query: SessionSearchQuery private backfill: SessionSearchBackfillState = 'idle' @@ -43,7 +52,7 @@ export class SessionSearchStore implements SessionSearchIndexSink { null private cleanupRequested = false private cleanup: Promise | null = null - private readonly maintenance: SessionSearchMaintenance + private warmed: Promise | null = null private lastIndexedAt: string | null = null private applyFailures = 0 private readonly stale = new Map() @@ -55,13 +64,13 @@ export class SessionSearchStore implements SessionSearchIndexSink { console.warn( '[ai-vault-search] index write failed:', error instanceof Error ? error.name : 'IndexError' - ) + ), + options: SessionSearchStoreOptions = {} ) { this.db = openSessionSearchDatabase(path) recoverSearchWrites(this.db) - this.writer = new SessionSearchIndexWriter(this.db) + this.writer = new SessionSearchIndexWriter(this.db, options.walBudgetBytes) this.query = new SessionSearchQuery(this.db) - this.maintenance = new SessionSearchMaintenance(this.db, () => this.closed, this.onError) } indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null { @@ -119,15 +128,13 @@ export class SessionSearchStore implements SessionSearchIndexSink { const epoch = this.writeEpoch const finish = this.indexing.beginWrite() try { - const applied = await this.writer.apply( - update, - () => + const applied = await this.writer.apply(update, { + active: () => !update.signal?.aborted && epoch === this.writeEpoch && this.acceptsCandidate(update.candidate), - undefined, - () => !this.closed - ) + available: () => !this.closed + }) if (!applied) { this.markStale(update.candidate) return @@ -164,10 +171,6 @@ export class SessionSearchStore implements SessionSearchIndexSink { return candidates } - get staleCount(): number { - return this.stale.size - } - removeFile(path: string): void { this.providerCounts = null this.stale.delete(path) @@ -221,7 +224,7 @@ export class SessionSearchStore implements SessionSearchIndexSink { } ) if (!this.closed && !signal?.aborted) { - await this.maintenance.compact(signal) + await compactSessionSearchIndex(this.db, () => this.closed || signal?.aborted === true) } } catch (error) { if (!this.closed) { @@ -231,7 +234,10 @@ export class SessionSearchStore implements SessionSearchIndexSink { } warm(): Promise { - return this.maintenance.warm() + this.warmed ??= warmSessionSearchPages(this.db, () => this.closed).catch((error) => + this.onError(error) + ) + return this.warmed } setBackfillState(state: SessionSearchBackfillState): void { @@ -253,7 +259,16 @@ export class SessionSearchStore implements SessionSearchIndexSink { const startedAt = performance.now() const execution = this.query.execute(args, aiVaultSearchHistoryCutoffMs(this.historyDays)) const durationMs = performance.now() - startedAt - this.maintenance.logQuery(args.query, execution.route, execution.hits.length, durationMs) + try { + logSessionSearchQuery(this.db, { + query: args.query, + route: execution.route, + hits: execution.hits.length, + durationMs + }) + } catch (error) { + this.onError(error) + } return { hits: execution.hits, route: execution.route, @@ -267,9 +282,7 @@ export class SessionSearchStore implements SessionSearchIndexSink { const providers = (this.providerCounts ??= this.db .prepare( `SELECT s.agent AS agent, COUNT(DISTINCT s.id) AS sessions, COUNT(m.id) AS messages - FROM sessions s LEFT JOIN messages m ON m.session_row_id = s.id - AND (m.batch_id IS NULL OR m.batch_id NOT IN (SELECT id FROM search_write_batches WHERE published=0)) - WHERE s.index_ready=1 AND s.id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL) + FROM ${VISIBLE_SESSIONS} s LEFT JOIN ${VISIBLE_MESSAGES} m ON m.session_row_id = s.id GROUP BY s.agent ORDER BY s.agent` ) .all() as { agent: AiVaultAgent; sessions: number; messages: number }[]) diff --git a/src/main/ai-vault-search/session-search-streaming-write.test.ts b/src/main/ai-vault-search/session-search-streaming-write.test.ts index eba9d88b0cd..c8713687203 100644 --- a/src/main/ai-vault-search/session-search-streaming-write.test.ts +++ b/src/main/ai-vault-search/session-search-streaming-write.test.ts @@ -6,13 +6,17 @@ import { import { captureIndexedSessionParse } from '../ai-vault/session-search-indexed-parse' import { SessionSearchIndexWriter } from './session-search-index-writer' import { SessionSearchStore } from './session-search-store' -import { stagedWriteUpdate } from './session-search-staged-write-fixtures' +import { + openSessionSearchIndexFile, + stagedWriteUpdate +} from './session-search-staged-write-test-fixture' it.each(['publish', 'reject', 'throw', 'cancel'] as const)( 'streams capture with atomic %s', async (outcome) => { - const store = new SessionSearchStore(':memory:') - const writer = new SessionSearchIndexWriter(store.db) + const index = await openSessionSearchIndexFile('ss-streaming') + const store = new SessionSearchStore(index.path) + const writer = new SessionSearchIndexWriter(index.db) const original = stagedWriteUpdate('oldneedle', 1) const replacement = stagedWriteUpdate('newneedle', 600) let active = true @@ -20,16 +24,10 @@ it.each(['publish', 'reject', 'throw', 'cancel'] as const)( await writer.apply(original) const parse = captureIndexedSessionParse( { - streamingCapture: true, indexedFile: () => null, markStale: () => {}, apply: async (update) => { - await writer.apply( - update, - () => active, - undefined, - () => true - ) + await writer.apply(update, { active: () => active, available: () => true }) } }, replacement, @@ -40,7 +38,7 @@ it.each(['publish', 'reject', 'throw', 'cancel'] as const)( if (i === 400) { expect( Number( - (store.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n + (index.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n ) ).toBeGreaterThan(128) expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0) @@ -76,18 +74,18 @@ it.each(['publish', 'reject', 'throw', 'cancel'] as const)( expect(store.search({ query: 'newneedle' }).hits[0].title).toBe('newneedle') } await store.purgeOlderThan(null) - expect( - store.db.prepare('SELECT id FROM search_write_batches WHERE published=0').all() - ).toHaveLength(0) + expect(index.db.prepare('SELECT id FROM search_write_batches').all()).toHaveLength(0) } finally { store.close() + await index.close() } } ) it('waits for final metadata after the last message without mutating the write', async () => { - const store = new SessionSearchStore(':memory:') - const writer = new SessionSearchIndexWriter(store.db) + const index = await openSessionSearchIndexFile('ss-streaming-final') + const store = new SessionSearchStore(index.path) + const writer = new SessionSearchIndexWriter(index.db) const replacement = stagedWriteUpdate('finaltitle', 1) const { promise: result, resolve } = Promise.withResolvers<{ session: typeof replacement.session @@ -117,5 +115,6 @@ it('waits for final metadata after the last message without mutating the write', } finally { resolve({ session: null, byteOffset: 0 }) store.close() + await index.close() } }) diff --git a/src/main/ai-vault-search/session-search-typo-policy.test.ts b/src/main/ai-vault-search/session-search-typo-policy.test.ts index 5682c0ab0f6..ba0524f33a2 100644 --- a/src/main/ai-vault-search/session-search-typo-policy.test.ts +++ b/src/main/ai-vault-search/session-search-typo-policy.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { SessionSearchStore } from './session-search-store' +import { openSessionSearchIndexFile } from './session-search-staged-write-test-fixture' import { SessionSearchTypoRepair } from './session-search-typo-repair' describe('typo repair policy', () => { @@ -12,19 +12,19 @@ describe('typo repair policy', () => { { input: 'calm', candidate: 'clam', copies: 2, exact: false, expected: null } ])( 'repairs $input to $expected with $copies postings (exact=$exact)', - ({ input, candidate, copies, exact, expected }) => { - const store = new SessionSearchStore(':memory:') + async ({ input, candidate, copies, exact, expected }) => { + const index = await openSessionSearchIndexFile('ss-typo-policy') try { - const insert = store.db.prepare('INSERT INTO messages_fts(user_text) VALUES (?)') + const insert = index.db.prepare('INSERT INTO messages_fts(user_text) VALUES (?)') for (let i = 0; i < copies; i++) { insert.run(candidate) } if (exact) { insert.run(input) } - expect(new SessionSearchTypoRepair(store.db).correct(input)).toBe(expected) + expect(new SessionSearchTypoRepair(index.db).correct(input)).toBe(expected) } finally { - store.close() + await index.close() } } ) diff --git a/src/main/ai-vault-search/session-search-typo-repair.ts b/src/main/ai-vault-search/session-search-typo-repair.ts index 27af7031280..9f5aa4ccae5 100644 --- a/src/main/ai-vault-search/session-search-typo-repair.ts +++ b/src/main/ai-vault-search/session-search-typo-repair.ts @@ -1,4 +1,6 @@ import type SyncDatabase from '../sqlite/sync-database' +import { quoteFtsTerm } from './session-search-query-planner' +import { VISIBLE_MESSAGES, VISIBLE_SESSIONS } from './session-search-schema' // Why: a query term with zero postings is usually a typo. The index's own // vocabulary (fts5vocab) is the dictionary, so repair needs no model and can @@ -43,13 +45,12 @@ export class SessionSearchTypoRepair { constructor(db: SyncDatabase) { this.unpublished = db.prepare( - 'SELECT 1 FROM search_write_batches WHERE published=0 UNION ALL SELECT 1 FROM search_pending_deletes LIMIT 1' + 'SELECT 1 FROM search_write_batches UNION ALL SELECT 1 FROM search_pending_deletes LIMIT 1' ) this.visiblePostings = - db.prepare(`SELECT m.id FROM messages_fts JOIN messages m ON m.id=messages_fts.rowid - JOIN sessions s ON s.id=m.session_row_id WHERE messages_fts MATCH ? AND s.index_ready=1 - AND s.id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL) - AND (m.batch_id IS NULL OR m.batch_id NOT IN (SELECT id FROM search_write_batches WHERE published=0)) LIMIT 2`) + db.prepare(`SELECT m.id FROM messages_fts JOIN ${VISIBLE_MESSAGES} m ON m.id=messages_fts.rowid + JOIN ${VISIBLE_SESSIONS} s ON s.id=m.session_row_id WHERE messages_fts MATCH ? + LIMIT ${MIN_DOC_FREQUENCY}`) this.exactMatch = db.prepare( 'SELECT rowid FROM messages_fts WHERE messages_fts MATCH ? LIMIT 1' @@ -65,17 +66,21 @@ export class SessionSearchTypoRepair { } hasPostings(term: string): boolean { - if (this.unpublished.get()) { - return this.visiblePostings.all(`"${term.replaceAll('"', '""')}"`).length > 0 + if (this.hasUnpublishedWrites()) { + return this.visiblePostings.all(quoteFtsTerm(term)).length > 0 } const row = this.documentFrequency.get(term.toLowerCase()) as VocabRow | undefined // unicode61 also folds Latin diacritics; raw vocabulary spelling alone can miss an exact hit. return ( - (row !== undefined && row.doc > 0) || - this.exactMatch.get(`"${term.replaceAll('"', '""')}"`) !== undefined + (row !== undefined && row.doc > 0) || this.exactMatch.get(quoteFtsTerm(term)) !== undefined ) } + /** A staged write is uncommitted, so `messages_vocab` can list a term no visible row has yet. */ + private hasUnpublishedWrites(): boolean { + return this.unpublished.get() !== undefined + } + /** Returns the closest indexed term, or null when `term` exists or nothing is close enough. */ correct(term: string): string | null { const lowered = term.toLowerCase() @@ -87,32 +92,40 @@ export class SessionSearchTypoRepair { } // Two-letter prefix first (a typo rarely hits both), then the transposed // pair, then the bare first letter as the wide fallback. - const unpublished = this.unpublished.get() !== undefined const prefixes = [lowered.slice(0, 2), lowered[1] + lowered[0], lowered[0]] - let best: { term: string; score: number; doc: number } | null = null for (const prefix of prefixes) { - for (const row of this.candidates(prefix, lowered.length)) { - const score = similarity(lowered, row.term) - if (score < MIN_SIMILARITY) { - continue - } - if ( - unpublished && - this.visiblePostings.all(`"${row.term.replaceAll('"', '""')}"`).length < MIN_DOC_FREQUENCY - ) { - continue - } - if (!best || score > best.score || (score === best.score && row.doc > best.doc)) { - best = { term: row.term, score, doc: row.doc } - } - } - if (best) { + const best = this.closest(lowered, prefix) + // Why the visibility probe is on the winner alone: ranking is pure CPU, + // but each probe is an FTS MATCH, and during a backfill — exactly when + // people search — one per candidate is thousands of queries per prefix. + if (best && this.isVisible(best.term)) { return best.term } } return null } + private closest(lowered: string, prefix: string): { term: string; score: number } | null { + let best: { term: string; score: number; doc: number } | null = null + for (const row of this.candidates(prefix, lowered.length)) { + const score = similarity(lowered, row.term) + if (score < MIN_SIMILARITY) { + continue + } + if (!best || score > best.score || (score === best.score && row.doc > best.doc)) { + best = { term: row.term, score, doc: row.doc } + } + } + return best + } + + private isVisible(term: string): boolean { + return ( + !this.hasUnpublishedWrites() || + this.visiblePostings.all(quoteFtsTerm(term)).length >= MIN_DOC_FREQUENCY + ) + } + private candidates(prefix: string, length: number): VocabRow[] { const last = prefix.charCodeAt(prefix.length - 1) const upper = prefix.slice(0, -1) + String.fromCharCode(last + 1) diff --git a/src/main/ai-vault-search/session-search-wal-budget.test.ts b/src/main/ai-vault-search/session-search-wal-budget.test.ts index 656c4ec98e9..df31aef3db0 100644 --- a/src/main/ai-vault-search/session-search-wal-budget.test.ts +++ b/src/main/ai-vault-search/session-search-wal-budget.test.ts @@ -1,65 +1,57 @@ -import { removeTree } from '../../shared/windows-transient-lock-removal' -import { mkdtemp } from 'node:fs/promises' -import { join } from 'node:path' -import { tmpdir } from 'node:os' -import { expect, it, vi } from 'vitest' -import * as Wal from './session-search-wal-budget' -import { stagedWriteUpdate } from './session-search-staged-write-fixtures' +import { expect, it } from 'vitest' import SyncDatabase from '../sqlite/sync-database' +import { + openSessionSearchIndexFile, + stagedWriteUpdate +} from './session-search-staged-write-test-fixture' import { SessionSearchStore } from './session-search-store' import { assertSearchWalBudget, SearchWalBackpressureError } from './session-search-wal-budget' it('backpressures a pinned snapshot and resumes checkpoints after that reader releases', async () => { - const root = await mkdtemp(join(tmpdir(), 'ss-wal-budget-')) - const path = join(root, 'index.sqlite') - const store = new SessionSearchStore(path) - const reader = new SyncDatabase(path, { readonly: true }) + const index = await openSessionSearchIndexFile('ss-wal-budget') + const reader = new SyncDatabase(index.path, { readonly: true }) try { - assertSearchWalBudget(store.db) + assertSearchWalBudget(index.db) reader.exec('BEGIN') reader.prepare('SELECT count(*) FROM sessions').get() - store.db + index.db .prepare("INSERT INTO search_log(ts,query,route,hits,duration_ms) VALUES ('t',?,'or',0,0)") .run('synthetic'.repeat(10000)) - expect(() => assertSearchWalBudget(store.db, 4096)).toThrow(SearchWalBackpressureError) + expect(() => assertSearchWalBudget(index.db, 4096)).toThrow(SearchWalBackpressureError) reader.exec('COMMIT') - expect(() => assertSearchWalBudget(store.db, 4096)).not.toThrow() + expect(() => assertSearchWalBudget(index.db, 4096)).not.toThrow() } finally { reader.close() - store.close() - await removeTree(root) + await index.close() } }) it('retains the old searchable generation and retries a backpressured write after reader release', async () => { - const root = await mkdtemp(join(tmpdir(), 'ss-wal-write-')) - const path = join(root, 'index.sqlite') + const index = await openSessionSearchIndexFile('ss-wal-write') const errors: unknown[] = [] - const store = new SessionSearchStore(path, (error) => errors.push(error)) - const reader = new SyncDatabase(path, { readonly: true }) + const store = new SessionSearchStore(index.path, (error) => errors.push(error), { + walBudgetBytes: 4096 + }) + const reader = new SyncDatabase(index.path, { readonly: true }) try { await store.apply(stagedWriteUpdate('oldneedle', 1)) reader.exec('BEGIN') reader.prepare('SELECT count(*) FROM messages').get() - const actual = Wal.assertSearchWalBudget - const spy = vi.spyOn(Wal, 'assertSearchWalBudget').mockImplementation((db) => actual(db, 4096)) await store.apply(stagedWriteUpdate('newneedle', 1000, 'append')) expect(errors.some((error) => error instanceof SearchWalBackpressureError)).toBe(true) - expect(store.staleCount).toBe(1) + expect(store.coverage().filesPending).toBe(1) expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1) expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0) expect(store.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(1) reader.exec('COMMIT') - spy.mockRestore() await store.apply(stagedWriteUpdate('newneedle', 1000, 'append')) await store.purgeOlderThan(null) expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1) - expect(store.staleCount).toBe(0) - expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1001 }) + expect(store.coverage().filesPending).toBe(0) + expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1001 }) } finally { - vi.restoreAllMocks() reader.close() store.close() - await removeTree(root) + await index.close() } }) diff --git a/src/main/ai-vault/session-newest-files.ts b/src/main/ai-vault/session-newest-files.ts index cb20a62b34f..c315c26dfad 100644 --- a/src/main/ai-vault/session-newest-files.ts +++ b/src/main/ai-vault/session-newest-files.ts @@ -7,10 +7,11 @@ export class SessionNewestFiles { private readonly limit: number constructor(limit: number) { - this.limit = limit === Infinity ? limit : Math.max(0, Math.trunc(limit) || 0) + this.limit = Math.max(0, Math.trunc(limit) || 0) } add(file: FileWithMtime): void { + // The backfill enumerates with no limit; skip the insert search entirely. if (!Number.isFinite(this.limit)) { this.files.push(file) return @@ -42,7 +43,8 @@ export class SessionNewestFiles { return this.files.length } + /** The unbounded path appends in traversal order, so the sort is not redundant. */ newest(): FileWithMtime[] { - return this.files.sort((a, b) => b.mtimeMs - a.mtimeMs) + return [...this.files].sort((a, b) => b.mtimeMs - a.mtimeMs) } } diff --git a/src/main/ai-vault/session-scanner-directory-reader.test.ts b/src/main/ai-vault/session-scanner-directory-reader.test.ts index 1a578f661c7..4ee0ded06f9 100644 --- a/src/main/ai-vault/session-scanner-directory-reader.test.ts +++ b/src/main/ai-vault/session-scanner-directory-reader.test.ts @@ -2,7 +2,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { mkdir, mkdtemp, readdir, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { walkSessionFiles } from './session-scanner-discovery' +import { forEachSessionFile, walkSessionFiles } from './session-scanner-discovery' let tempRoot: string | null = null @@ -76,13 +76,14 @@ it('visits file contents before descending further without retaining paths', asy a.name.localeCompare(b.name) ) }) - const retained = await walkSessionFiles(tempRoot, 'claude', [], { - extensions: new Set(['.jsonl']), - readDirectory, - onFile: async (path) => { + await forEachSessionFile( + tempRoot, + 'claude', + [], + { extensions: new Set(['.jsonl']), readDirectory }, + async (path) => { visited.push(path) } - }) - expect(retained).toEqual([]) + ) expect(visited).toHaveLength(2) }) diff --git a/src/main/ai-vault/session-scanner-discovery.ts b/src/main/ai-vault/session-scanner-discovery.ts index e152bc501ef..b9630a59b67 100644 --- a/src/main/ai-vault/session-scanner-discovery.ts +++ b/src/main/ai-vault/session-scanner-discovery.ts @@ -21,12 +21,17 @@ export async function discoverFiles(args: { }): Promise { const files = new SessionNewestFiles(args.limit) try { - await walkSessionFiles(args.rootDir, args.agent, args.issues, { - extensions: new Set(args.extensions), - signal: args.signal, - filePredicate: args.filePredicate, - directoryPredicate: args.directoryPredicate, - onFile: async (path) => { + await forEachSessionFile( + args.rootDir, + args.agent, + args.issues, + { + extensions: new Set(args.extensions), + signal: args.signal, + filePredicate: args.filePredicate, + directoryPredicate: args.directoryPredicate + }, + async (path) => { args.signal?.throwIfAborted() try { const fileStat = await wslGatedStat(path, 'scan') @@ -53,8 +58,11 @@ export async function discoverFiles(args: { }) } } - }) + ) } catch (err) { + // Why: discoverAiVaultSessionSources fans out with Promise.all, so one + // stalled distro would otherwise reject the whole vault scan — including + // every healthy local agent. Contain it to this root. if (!(err instanceof WslTranscriptFsError)) { throw err } @@ -85,22 +93,39 @@ async function optionalContentDependencyStat( } } +export type SessionFileWalkOptions = { + extensions: Set + filePredicate?: (path: string) => boolean + // Return false to skip descending into a directory; depth 0 is a child of + // rootDir, so pruned subtrees are never stat'd or parsed. + directoryPredicate?: (name: string, depth: number) => boolean + readDirectory?: (dirPath: string) => Promise + signal?: AbortSignal +} + +/** Collecting form for callers that want every match; large scans use `forEachSessionFile`. */ export async function walkSessionFiles( dirPath: string, agent: AiVaultAgent, issues: AiVaultScanIssue[], - options: { - extensions: Set - filePredicate?: (path: string) => boolean - // Return false to skip descending into a directory; depth 0 is a child of - // rootDir, so pruned subtrees are never stat'd or parsed. - directoryPredicate?: (name: string, depth: number) => boolean - readDirectory?: (dirPath: string) => Promise - signal?: AbortSignal - onFile?: (path: string) => Promise - }, - depth = 0 + options: SessionFileWalkOptions ): Promise { + const files: string[] = [] + await forEachSessionFile(dirPath, agent, issues, options, async (path) => { + files.push(path) + }) + return files +} + +/** Streams matches to `onFile` so a bounded consumer never retains the whole tree. */ +export async function forEachSessionFile( + dirPath: string, + agent: AiVaultAgent, + issues: AiVaultScanIssue[], + options: SessionFileWalkOptions, + onFile: (path: string) => Promise, + depth = 0 +): Promise { options.signal?.throwIfAborted() let entries try { @@ -114,10 +139,9 @@ export async function walkSessionFiles( if (error instanceof WslTranscriptFsError) { throw error } - return [] + return } - const files: string[] = [] for (const entry of entries) { options.signal?.throwIfAborted() const fullPath = join(dirPath, entry.name) @@ -125,7 +149,7 @@ export async function walkSessionFiles( // Skip whole subtrees an agent never wants (e.g. subagent transcripts), // avoiding the readdir cost of descending into them. if (options.directoryPredicate?.(entry.name, depth) ?? true) { - files.push(...(await walkSessionFiles(fullPath, agent, issues, options, depth + 1))) + await forEachSessionFile(fullPath, agent, issues, options, onFile, depth + 1) } continue } @@ -134,12 +158,7 @@ export async function walkSessionFiles( options.extensions.has(extname(entry.name).toLowerCase()) && (options.filePredicate?.(fullPath) ?? true) ) { - if (options.onFile) { - await options.onFile(fullPath) - } else { - files.push(fullPath) - } + await onFile(fullPath) } } - return files } diff --git a/src/main/ai-vault/session-scanner-jsonl-reader.ts b/src/main/ai-vault/session-scanner-jsonl-reader.ts index 15900fd5a83..62b04fd5731 100644 --- a/src/main/ai-vault/session-scanner-jsonl-reader.ts +++ b/src/main/ai-vault/session-scanner-jsonl-reader.ts @@ -62,10 +62,7 @@ export async function consumeCompleteJsonlLines(args: { } newlineIndex = data.indexOf(NEWLINE_BYTE, lineStart) } - const checkpoint = checkpointSessionSearchCapture() - if (checkpoint) { - await checkpoint - } + await checkpointSessionSearchCapture() consumedThrough += lineStart if (stopped) { remainderParts = [] diff --git a/src/main/ai-vault/session-scanner-opencode-pending-request.ts b/src/main/ai-vault/session-scanner-opencode-pending-request.ts new file mode 100644 index 00000000000..6d4cc2cc75b --- /dev/null +++ b/src/main/ai-vault/session-scanner-opencode-pending-request.ts @@ -0,0 +1,40 @@ +import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation' +import type { OpenCodeSqliteWorkerRequest } from './session-scanner-opencode-sqlite-worker-protocol' +import { getSessionSearchCaptureSignal } from './session-search-capture' +import type { OpenCodeCaptureConsumer } from './session-search-opencode-capture-channel' + +/** One request the client has accepted: queued or active, with its own deadline. */ +export type OpenCodePendingCall = { + request: OpenCodeSqliteWorkerRequest + timeoutMs: number + resolve: (value: unknown) => void + reject: (error: Error) => void + timer: NodeJS.Timeout | null + capture?: OpenCodeCaptureConsumer +} + +/** + * The request owns its abort listener until either queued or active work settles. + * Applies to every request kind, not just capturing parses: a backfill abort + * cancels the `list` legs it queued too. + */ +export function bindOpenCodeRequestCancellation( + resolve: (value: unknown) => void, + reject: (error: Error) => void, + cancel: () => void +): { resolve: typeof resolve; reject: typeof reject } { + const signal = getSessionSearchCaptureSignal() + throwIfAiVaultScanCancelled(signal) + signal?.addEventListener('abort', cancel, { once: true }) + const cleanup = (): void => signal?.removeEventListener('abort', cancel) + return { + resolve: (value) => { + cleanup() + resolve(value) + }, + reject: (error) => { + cleanup() + reject(error) + } + } +} diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite-schema.ts b/src/main/ai-vault/session-scanner-opencode-sqlite-schema.ts new file mode 100644 index 00000000000..b7420c80043 --- /dev/null +++ b/src/main/ai-vault/session-scanner-opencode-sqlite-schema.ts @@ -0,0 +1,29 @@ +import type SyncDatabase from '../sqlite/sync-database' +import { columnExists, tableExists } from '../opencode-usage/schema-helpers' + +// Why: OpenCode's schema has moved more than once, so every read probes for the +// columns it names. These are the two shapes the session parser depends on; +// keeping them here stops each reader from inventing its own partial gate. + +/** Enough of `message` to count a session's turns. */ +export function canCountOpenCodeMessages(db: SyncDatabase): boolean { + return ( + tableExists(db, 'message') && + columnExists(db, 'message', 'session_id') && + columnExists(db, 'message', 'data') + ) +} + +/** Enough of `message`×`part` to read a session's parts in turn order. */ +export function canReadOpenCodeMessageParts(db: SyncDatabase): boolean { + return ( + canCountOpenCodeMessages(db) && + columnExists(db, 'message', 'id') && + // Every parts read orders by it; unprobed, a schema without it throws mid-read. + columnExists(db, 'message', 'time_created') && + tableExists(db, 'part') && + columnExists(db, 'part', 'message_id') && + columnExists(db, 'part', 'time_created') && + columnExists(db, 'part', 'data') + ) +} diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.test.ts b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.test.ts index 03ce70c4114..df4677c721b 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.test.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.test.ts @@ -1,12 +1,12 @@ import type { Worker } from 'node:worker_threads' import { describe, expect, it, vi } from 'vitest' import { - IDLE_TEARDOWN_MS, LIST_TIMEOUT_MS, MAX_CONSECUTIVE_DEATHS, OpenCodeSqliteWorkerClient, PARSE_TIMEOUT_MS } from './session-scanner-opencode-sqlite-worker-client' +import { IDLE_TEARDOWN_MS } from './session-scanner-opencode-worker-host' import type { OpenCodeSqliteParseValue, OpenCodeSqliteParentMessage, @@ -14,9 +14,9 @@ import type { } from './session-scanner-opencode-sqlite-worker-protocol' import { isSessionSearchCaptureActive, - withSessionSearchCapture, withStreamingSessionSearchCapture } from './session-search-capture' +import type { SessionSearchCapturedMessage } from './session-search-capture' import type { AiVaultScanIssue, AiVaultSession } from '../../shared/ai-vault-types' // A worker_threads stand-in the tests drive directly: it records posted requests @@ -99,12 +99,12 @@ describe('OpenCodeSqliteWorkerClient', () => { // A response for a different id must not settle the active call. worker!.emit('message', { id: 999, - ok: true, + kind: 'result', value: null } satisfies OpenCodeSqliteWorkerResponse) worker!.emit('message', { id: worker!.lastId(), - ok: true, + kind: 'result', value: { session: { sessionId: 'a' } } } satisfies OpenCodeSqliteWorkerResponse) @@ -123,12 +123,20 @@ describe('OpenCodeSqliteWorkerClient', () => { expect(worker.postedRequests).toHaveLength(1) expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', sessionId: 'a' }) - worker.emit('message', { id: worker.postedRequests[0]!.id, ok: true, value: parseValue('A') }) + worker.emit('message', { + id: worker.postedRequests[0]!.id, + kind: 'result', + value: parseValue('A') + }) await first expect(worker.postedRequests).toHaveLength(2) expect(worker.postedRequests[1]).toMatchObject({ kind: 'parse', sessionId: 'b' }) - worker.emit('message', { id: worker.postedRequests[1]!.id, ok: true, value: parseValue('B') }) + worker.emit('message', { + id: worker.postedRequests[1]!.id, + kind: 'result', + value: parseValue('B') + }) await expect(second).resolves.toBe('B') // The worker is reused across serial calls (one persistent worker). expect(workers).toHaveLength(1) @@ -157,7 +165,7 @@ describe('OpenCodeSqliteWorkerClient', () => { const respawned = workers[1]! expect(respawned.postedRequests).toHaveLength(1) expect(respawned.postedRequests[0]).toMatchObject({ sessionId: 'b' }) - respawned.emit('message', { id: respawned.lastId(), ok: true, value: parseValue('B') }) + respawned.emit('message', { id: respawned.lastId(), kind: 'result', value: parseValue('B') }) await expect(queued).resolves.toBe('B') } finally { vi.useRealTimers() @@ -179,7 +187,7 @@ describe('OpenCodeSqliteWorkerClient', () => { // Exactly one respawn; the queued call drains on the new worker. expect(workers).toHaveLength(2) const respawned = workers[1]! - respawned.emit('message', { id: respawned.lastId(), ok: true, value: parseValue('B') }) + respawned.emit('message', { id: respawned.lastId(), kind: 'result', value: parseValue('B') }) await expect(queued).resolves.toBe('B') }) @@ -285,7 +293,7 @@ describe('OpenCodeSqliteWorkerClient', () => { expect(worker).toBeDefined() worker!.emit('message', { id: worker!.lastId(), - ok: true, + kind: 'result', value: { candidates: [], issues: [] } } satisfies OpenCodeSqliteWorkerResponse) await expect(secondPromise).resolves.toEqual([]) @@ -297,13 +305,21 @@ describe('OpenCodeSqliteWorkerClient', () => { const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) const first = client.parse({ dbPath: '/db#a', sessionId: 'a', platform: 'darwin' }) - workers[0]!.emit('message', { id: workers[0]!.lastId(), ok: true, value: parseValue('A') }) + workers[0]!.emit('message', { + id: workers[0]!.lastId(), + kind: 'result', + value: parseValue('A') + }) await expect(first).resolves.toBe('A') workers[0]!.emit('exit', 0) const second = client.parse({ dbPath: '/db#b', sessionId: 'b', platform: 'darwin' }) expect(workers).toHaveLength(2) - workers[1]!.emit('message', { id: workers[1]!.lastId(), ok: true, value: parseValue('B') }) + workers[1]!.emit('message', { + id: workers[1]!.lastId(), + kind: 'result', + value: parseValue('B') + }) await expect(second).resolves.toBe('B') }) @@ -317,14 +333,22 @@ describe('OpenCodeSqliteWorkerClient', () => { }) const first = client.parse({ dbPath: '/db#a', sessionId: 'a', platform: 'darwin' }) - workers[0]!.emit('message', { id: workers[0]!.lastId(), ok: true, value: parseValue('A') }) + workers[0]!.emit('message', { + id: workers[0]!.lastId(), + kind: 'result', + value: parseValue('A') + }) await expect(first).resolves.toBe('A') await vi.advanceTimersByTimeAsync(IDLE_TEARDOWN_MS) expect(workers[0]!.terminated).toBe(true) const second = client.parse({ dbPath: '/db#b', sessionId: 'b', platform: 'darwin' }) expect(workers).toHaveLength(2) - workers[1]!.emit('message', { id: workers[1]!.lastId(), ok: true, value: parseValue('B') }) + workers[1]!.emit('message', { + id: workers[1]!.lastId(), + kind: 'result', + value: parseValue('B') + }) await expect(second).resolves.toBe('B') } finally { vi.useRealTimers() @@ -364,7 +388,7 @@ describe('OpenCodeSqliteWorkerClient', () => { for (let i = 0; i < 3; i++) { const promise = client.parse({ dbPath: `/db#${i}`, sessionId: `s${i}`, platform: 'darwin' }) const worker = workers[0]! - worker.emit('message', { id: worker.lastId(), ok: true, value: parseValue(`v${i}`) }) + worker.emit('message', { id: worker.lastId(), kind: 'result', value: parseValue(`v${i}`) }) await expect(promise).resolves.toBe(`v${i}`) } expect(workers).toHaveLength(1) @@ -376,30 +400,34 @@ describe('OpenCodeSqliteWorkerClient search capture', () => { const workers: FakeWorker[] = [] const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) - const captured = await withSessionSearchCapture(async () => { - const parsePromise = client.parse({ dbPath: '/db', sessionId: 'a', platform: 'darwin' }) - const worker = workers[0]! - await vi.waitFor(() => expect(worker.postedRequests).toHaveLength(1)) - expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', capture: true }) - worker.emit('message', { - id: worker.lastId(), - ok: true, - captureBatch: 1, - value: [{ role: 'user', text: 'ballast tanks', timestamp: null }] - } satisfies OpenCodeSqliteWorkerResponse) - await vi.waitFor(() => - expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 }) - ) - worker.emit('message', { - id: worker.lastId(), - ok: true, - value: { session: { sessionId: 'a' } } - }) - return parsePromise - }) + const messages: SessionSearchCapturedMessage[] = [] + const value = await withStreamingSessionSearchCapture( + { push: (message) => messages.push(message), checkpoint: async () => {} }, + async () => { + const parsePromise = client.parse({ dbPath: '/db', sessionId: 'a', platform: 'darwin' }) + const worker = workers[0]! + await vi.waitFor(() => expect(worker.postedRequests).toHaveLength(1)) + expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', capture: true }) + worker.emit('message', { + id: worker.lastId(), + kind: 'batch', + batch: 1, + messages: [{ role: 'user', text: 'ballast tanks', timestamp: null }] + } satisfies OpenCodeSqliteWorkerResponse) + await vi.waitFor(() => + expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 }) + ) + worker.emit('message', { + id: worker.lastId(), + kind: 'result', + value: { session: { sessionId: 'a' } } + }) + return parsePromise + } + ) - expect(captured.value).toEqual({ sessionId: 'a' }) - expect(captured.messages).toEqual([{ role: 'user', text: 'ballast tanks', timestamp: null }]) + expect(value).toEqual({ sessionId: 'a' }) + expect(messages).toEqual([{ role: 'user', text: 'ballast tanks', timestamp: null }]) }) it('does not ask for capture outside a capture scope', async () => { @@ -412,7 +440,7 @@ describe('OpenCodeSqliteWorkerClient search capture', () => { expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', capture: false }) worker.emit('message', { id: worker.lastId(), - ok: true, + kind: 'result', value: { session: null } } satisfies OpenCodeSqliteWorkerResponse) @@ -435,9 +463,9 @@ it('acknowledges capture only after the caller channel drains, including across const worker = workers[0]! worker.emit('message', { id: worker.lastId(), - ok: true, - captureBatch: 1, - value: [{ role: 'user', text: 'late marker', timestamp: null }] + kind: 'batch', + batch: 1, + messages: [{ role: 'user', text: 'late marker', timestamp: null }] }) await Promise.resolve() expect(push).toHaveBeenCalledOnce() @@ -447,14 +475,16 @@ it('acknowledges capture only after the caller channel drains, including across await vi.waitFor(() => expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 }) ) - worker.emit('message', { id: worker.lastId(), ok: true, value: { session: { sessionId: 'a' } } }) + worker.emit('message', { + id: worker.lastId(), + kind: 'result', + value: { session: { sessionId: 'a' } } + }) expect(await parsed).toEqual({ sessionId: 'a' }) }) -it('rejects the parse and retires its worker when the capture consumer fails', async () => { - const workers: FakeWorker[] = [] - const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) - const parsed = withStreamingSessionSearchCapture( +function failingCaptureParse(client: OpenCodeSqliteWorkerClient): Promise { + return withStreamingSessionSearchCapture( { push() {}, checkpoint: async () => { @@ -463,37 +493,110 @@ it('rejects the parse and retires its worker when the capture consumer fails', a }, () => client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform }) ) +} + +it('rejects the parse and retires its worker when the capture consumer fails', async () => { + const workers: FakeWorker[] = [] + const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) + const parsed = failingCaptureParse(client) const rejected = expect(parsed).rejects.toThrow('capture stopped') - workers[0]!.emit('message', { id: workers[0]!.lastId(), ok: true, captureBatch: 1, value: [] }) + workers[0]!.emit('message', { id: workers[0]!.lastId(), kind: 'batch', batch: 1, messages: [] }) await rejected + // The worker is parked on an ack that will never arrive, so it has to go. expect(workers[0]!.terminated).toBe(true) expect(workers[0]!.postedRequests).toHaveLength(1) }) -it('suspends the producer timeout during backpressure and restores it after acknowledgement', async () => { +it('does not count a failing index write as a worker death', async () => { + const workers: FakeWorker[] = [] + const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) + + // One burst: enough capturing parses to hit the respawn cap, plus an unrelated + // list for another database queued behind them. The death counter only resets + // from full idle, so nothing here hides an increment. + const parses = withStreamingSessionSearchCapture( + { + push() {}, + checkpoint: async () => { + throw new Error('capture stopped') + } + }, + async () => + Array.from({ length: MAX_CONSECUTIVE_DEATHS }, (_, i) => + client.parse({ dbPath: `/db#${i}`, sessionId: `s${i}`, platform: process.platform }) + ) + ) + const issues: AiVaultScanIssue[] = [] + const list = client.list({ dbPaths: ['/db#other'], limit: 10, issues }) + + for (const parse of await parses) { + const rejected = expect(parse).rejects.toThrow('capture stopped') + const worker = workers.at(-1)! + worker.emit('message', { id: worker.lastId(), kind: 'batch', batch: 1, messages: [] }) + await rejected + } + + // A healthy worker parked on an unreachable ack is not a crash, so the queued + // list must still be dispatched rather than drained as a crash loop. + const worker = workers.at(-1)! + expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'list' }) + worker.emit('message', { + id: worker.lastId(), + kind: 'result', + value: { candidates: [], issues: [] } + }) + await expect(list).resolves.toEqual([]) + expect(issues).toEqual([]) +}) + +it('caps a single backpressure stall at the parse deadline instead of waiting forever', async () => { vi.useFakeTimers() try { const workers: FakeWorker[] = [] const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) - let release!: () => void - const drained = new Promise((resolve) => { - release = resolve - }) - const parsed = withStreamingSessionSearchCapture({ push() {}, checkpoint: () => drained }, () => - client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform }) + // A consumer that never resolves stands in for a wedged index writer. + const parsed = withStreamingSessionSearchCapture( + { push() {}, checkpoint: () => new Promise(() => {}) }, + () => client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform }) ) const failure = expect(parsed).rejects.toThrow('timed out') const worker = workers[0]! - worker.emit('message', { id: worker.lastId(), ok: true, captureBatch: 1, value: [] }) - await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS + 1) + worker.emit('message', { id: worker.lastId(), kind: 'batch', batch: 1, messages: [] }) + + // The batch restarts the deadline rather than removing it. + await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS - 1) expect(worker.terminated).toBe(false) - release() - await vi.advanceTimersByTimeAsync(0) - expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 }) - await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS) + await vi.advanceTimersByTimeAsync(2) await failure expect(worker.terminated).toBe(true) } finally { vi.useRealTimers() } }) + +it('lets total production run past the deadline as long as batches keep arriving', async () => { + vi.useFakeTimers() + try { + const workers: FakeWorker[] = [] + const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} }) + const parsed = withStreamingSessionSearchCapture( + { push() {}, checkpoint: async () => {} }, + () => client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform }) + ) + const worker = workers[0]! + + // Four batches, each landing just under the deadline: cumulative time is far + // past PARSE_TIMEOUT_MS, but no single gap is, so the parse must survive. + for (let batch = 1; batch <= 4; batch++) { + worker.emit('message', { id: worker.lastId(), kind: 'batch', batch, messages: [] }) + await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS - 1) + expect(worker.terminated).toBe(false) + } + expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 4 }) + + worker.emit('message', { id: worker.lastId(), kind: 'result', value: parseValue('A') }) + await expect(parsed).resolves.toBe('A') + } finally { + vi.useRealTimers() + } +}) diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.ts b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.ts index 339a9df92e4..46382272236 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-client.ts @@ -1,4 +1,3 @@ -import type { Worker } from 'node:worker_threads' import type { AiVaultScanIssue, AiVaultSession } from '../../shared/ai-vault-types' import type { OpenCodeSqliteListValue, @@ -9,11 +8,17 @@ import type { import { createAiVaultScanCancelledError } from './ai-vault-scan-cancellation' import { isSessionSearchCaptureActive } from './session-search-capture' import { - bindOpenCodeCaptureConsumer, - bindOpenCodeCaptureCancellation, - receiveOpenCodeCaptureBatch, + bindOpenCodeRequestCancellation, type OpenCodePendingCall as PendingCall -} from './session-search-opencode-worker-receiver' +} from './session-scanner-opencode-pending-request' +import { + bindOpenCodeCaptureConsumer, + receiveOpenCodeCaptureBatch +} from './session-search-opencode-capture-channel' +import { + OpenCodeSqliteWorkerHost, + type WorkerFactory +} from './session-scanner-opencode-worker-host' import type { SessionFileCandidate } from './session-scanner-types' import { errorMessage } from './session-scanner-values' @@ -25,15 +30,12 @@ import { errorMessage } from './session-scanner-values' export const LIST_TIMEOUT_MS = 30_000 export const PARSE_TIMEOUT_MS = 15_000 -export const IDLE_TEARDOWN_MS = 30_000 // After this many consecutive worker deaths, fail the remaining queued calls to // scan issues instead of respawning so a DB that reliably kills the worker can't // spin a crash loop. Reset on any successful response, after draining, and when a // fresh scan burst starts from idle (so the cap is per-scan, not process-wide). export const MAX_CONSECUTIVE_DEATHS = 3 -export type WorkerFactory = () => Worker - // Distinguishes "no worker available at all" from a timeout or crash so callers // can surface a precise issue while keeping synchronous SQLite off the main thread. class OpenCodeSqliteWorkerUnavailableError extends Error {} @@ -46,20 +48,21 @@ class OpenCodeSqliteWorkerUnavailableError extends Error {} * no worker can be spawned rather than moving SQLite work onto the main thread. */ export class OpenCodeSqliteWorkerClient { - private worker: Worker | null = null private active: PendingCall | null = null private queue: PendingCall[] = [] - private idleTimer: NodeJS.Timeout | null = null private consecutiveDeaths = 0 private nextId = 1 - private loggedWorkerUnavailable = false - private cleanupWorkerListeners: (() => void) | null = null - private readonly workerFactory: WorkerFactory - private readonly log: (message: string) => void + private readonly host: OpenCodeSqliteWorkerHost constructor(options: { workerFactory: WorkerFactory; log?: (message: string) => void }) { - this.workerFactory = options.workerFactory - this.log = options.log ?? ((message) => console.warn(message)) + this.host = new OpenCodeSqliteWorkerHost({ + factory: options.workerFactory, + log: options.log ?? ((message) => console.warn(message)), + onMessage: (response) => this.onMessage(response), + onError: (error) => this.onWorkerFault(error), + onExit: (code) => this.onWorkerExit(code), + isIdle: () => !this.active && this.queue.length === 0 + }) } /** @@ -160,7 +163,7 @@ export class OpenCodeSqliteWorkerClient { const call: PendingCall = { request: { ...request, id } as PendingCall['request'], timeoutMs, - ...bindOpenCodeCaptureCancellation(resolve, reject, () => this.cancel(call)), + ...bindOpenCodeRequestCancellation(resolve, reject, () => this.cancel(call)), timer: null, capture } @@ -171,7 +174,7 @@ export class OpenCodeSqliteWorkerClient { private cancel(call: PendingCall): void { if (this.active === call) { - this.destroyWorker() + this.host.destroy() } this.queue = this.queue.filter((pending) => pending !== call) this.settle(call, () => call.reject(createAiVaultScanCancelledError())) @@ -182,7 +185,7 @@ export class OpenCodeSqliteWorkerClient { if (this.active || this.queue.length === 0) { return } - const worker = this.ensureWorker() + const worker = this.host.ensure() if (!worker) { this.failQueuedAsUnavailable() return @@ -192,7 +195,7 @@ export class OpenCodeSqliteWorkerClient { return } this.active = call - this.clearIdleTimer() + this.host.clearIdleTimer() // Timeout clock starts at dispatch (not enqueue): a batch may enqueue up to // 8 parses at once, and a queue-inclusive timeout would fire falsely. call.timer = setTimeout(() => this.onTimeout(call), call.timeoutMs) @@ -200,61 +203,38 @@ export class OpenCodeSqliteWorkerClient { worker.postMessage(call.request) } - private ensureWorker(): Worker | null { - if (this.worker) { - return this.worker - } - try { - const worker = this.workerFactory() - const onMessage = (response: OpenCodeSqliteWorkerResponse): void => this.onMessage(response) - const onError = (error: Error): void => this.onWorkerFault(error) - const onExit = (code: number): void => this.onWorkerExit(code) - worker.on('message', onMessage) - worker.on('error', onError) - worker.on('exit', onExit) - this.cleanupWorkerListeners = () => { - worker.off('message', onMessage) - worker.off('error', onError) - worker.off('exit', onExit) - } - // Never keep the app alive for a scan worker. - worker.unref?.() - this.worker = worker - return worker - } catch (err) { - // Why (#8864): never fall back to synchronous SQLite reads here; a missing - // bundle or resource-exhausted spawn must omit OpenCode history rather than - // reintroduce the main-process hang this worker boundary prevents. - if (!this.loggedWorkerUnavailable) { - this.loggedWorkerUnavailable = true - this.log(`OpenCode SQLite worker unavailable; skipping its history. ${errorMessage(err)}`) - } - return null - } - } - private onMessage(response: OpenCodeSqliteWorkerResponse): void { const call = this.active if (!call || call.request.id !== response.id) { return } - if (response.ok && response.captureBatch !== undefined) { + if (response.kind === 'batch') { receiveOpenCodeCaptureBatch({ call, - response, - worker: this.worker, + batch: response, + worker: this.host.current, isActive: () => this.active === call, onTimeout: () => this.onTimeout(call), - onError: (error) => this.onWorkerFault(error) + onProtocolViolation: (error) => this.onWorkerFault(error), + onConsumerError: (error) => this.onCaptureConsumerFailure(call, error) }) return } this.consecutiveDeaths = 0 - if (response.ok) { - this.settle(call, () => call.resolve(response.value)) - } else { - this.settle(call, () => call.reject(new Error(response.error))) - } + this.settle(call, () => + response.kind === 'result' + ? call.resolve(response.value) + : call.reject(new Error(response.error)) + ) + this.afterSettle() + } + + // The worker is healthy, only parked on an ack that will never arrive, so it + // is retired without counting a death: a failing index write must not spend + // the respawn budget that unrelated queued calls depend on. + private onCaptureConsumerFailure(call: PendingCall, error: Error): void { + this.host.destroy() + this.settle(call, () => call.reject(error)) this.afterSettle() } @@ -269,7 +249,7 @@ export class OpenCodeSqliteWorkerClient { // A clean self-exit is not a death, but the stale handle must be dropped // or the next dispatch would post into the dead worker and stall to timeout. if (code === 0 && !this.active && this.queue.length === 0) { - this.destroyWorker() + this.host.destroy() return } this.onWorkerFault(new Error(`OpenCode SQLite worker exited with code ${code}`)) @@ -277,7 +257,7 @@ export class OpenCodeSqliteWorkerClient { private onWorkerFault(error: Error): void { const failed = this.active - this.destroyWorker() + this.host.destroy() this.consecutiveDeaths++ if (failed) { this.settle(failed, () => failed.reject(error)) @@ -328,46 +308,7 @@ export class OpenCodeSqliteWorkerClient { if (this.queue.length > 0) { this.pump() } else { - this.scheduleIdleTeardown() + this.host.scheduleIdleTeardown() } } - - private scheduleIdleTeardown(): void { - this.clearIdleTimer() - if (!this.worker) { - return - } - this.idleTimer = setTimeout(() => this.teardownIfIdle(), IDLE_TEARDOWN_MS) - this.idleTimer.unref?.() - } - - private teardownIfIdle(): void { - this.idleTimer = null - // Only tear down with nothing active AND nothing queued: a request arriving - // as the timer fires must never be lost to a self-exiting worker. - if (this.active || this.queue.length > 0) { - return - } - this.destroyWorker() - } - - private clearIdleTimer(): void { - if (this.idleTimer) { - clearTimeout(this.idleTimer) - this.idleTimer = null - } - } - - private destroyWorker(): void { - this.clearIdleTimer() - const worker = this.worker - this.worker = null - if (!worker) { - return - } - this.cleanupWorkerListeners?.() - this.cleanupWorkerListeners = null - worker.removeAllListeners() - void worker.terminate().catch(() => undefined) - } } diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.test.ts b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.test.ts index 5de800b0581..935d25ae38f 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.test.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.test.ts @@ -32,9 +32,9 @@ vi.mock('node:worker_threads', () => ({ }, postMessage(response: OpenCodeSqliteWorkerResponse) { posted.push(response) - if (acknowledge && response.ok && response.captureBatch !== undefined) { + if (acknowledge && response.kind === 'batch') { queueMicrotask(() => - handler?.({ id: response.id, kind: 'captureAck', batch: response.captureBatch! }) + handler?.({ id: response.id, kind: 'captureAck', batch: response.batch }) ) } } @@ -89,13 +89,14 @@ function createDbWithOneTurn(): string { async function parseOnWorker(dbPath: string, capture: boolean): Promise { handler?.({ id: 1, kind: 'parse', dbPath, sessionId: SESSION_ID, platform: 'darwin', capture }) - await vi.waitFor(() => - expect(posted.some((reply) => !reply.ok || reply.captureBatch === undefined)).toBe(true) - ) + await vi.waitFor(() => expect(posted.some((reply) => reply.kind !== 'batch')).toBe(true)) const response = posted.at(-1)! - if (!response.ok) { + if (response.kind === 'error') { throw new Error(response.error) } + if (response.kind !== 'result') { + throw new Error('worker replied with a capture batch instead of a result') + } return response.value as OpenCodeSqliteParseValue } @@ -104,11 +105,7 @@ describe('OpenCode SQLite worker entry', () => { const value = await parseOnWorker(createDbWithOneTurn(), true) expect(value.session?.sessionId).toBe(SESSION_ID) - expect( - posted - .filter((reply) => reply.ok && reply.captureBatch !== undefined) - .flatMap((reply) => (reply.ok ? reply.value : [])) - ).toEqual([ + expect(posted.flatMap((reply) => (reply.kind === 'batch' ? reply.messages : []))).toEqual([ { role: 'user', text: 'recalibrate the ballast pump', timestamp: expect.any(String) } ]) }) @@ -117,11 +114,7 @@ describe('OpenCode SQLite worker entry', () => { const value = await parseOnWorker(createDbWithOneTurn(), false) expect(value.session?.sessionId).toBe(SESSION_ID) - expect( - posted - .filter((reply) => reply.ok && reply.captureBatch !== undefined) - .flatMap((reply) => (reply.ok ? reply.value : [])) - ).toEqual([]) + expect(posted.flatMap((reply) => (reply.kind === 'batch' ? reply.messages : []))).toEqual([]) }) }) @@ -153,17 +146,18 @@ it('waits for downstream acknowledgement between bounded batches without droppin }) await vi.waitFor(() => expect(posted).toHaveLength(1)) const first = posted[0]! - expect(first).toMatchObject({ ok: true, captureBatch: 1 }) + expect(first).toMatchObject({ kind: 'batch', batch: 1 }) await new Promise((resolve) => setTimeout(resolve, 20)) expect(posted).toHaveLength(1) acknowledge = true handler?.({ id: 2, kind: 'captureAck', batch: 1 }) await vi.waitFor(() => - expect(posted.at(-1)).toMatchObject({ ok: true, value: { session: { sessionId: SESSION_ID } } }) - ) - const batches = posted.flatMap((reply) => - reply.ok && reply.captureBatch !== undefined ? [reply.value as { text: string }[]] : [] + expect(posted.at(-1)).toMatchObject({ + kind: 'result', + value: { session: { sessionId: SESSION_ID } } + }) ) + const batches = posted.flatMap((reply) => (reply.kind === 'batch' ? [reply.messages] : [])) expect(batches.length).toBeGreaterThan(2) expect(batches.flat()).toHaveLength(count + 1) expect(batches.flat().at(-1)?.text).toBe(`${count - 1} ${text}`.trim()) diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.ts b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.ts index ff4a771f8e3..77b787b36a4 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-entry.ts @@ -33,11 +33,15 @@ async function handleRequest( limit: request.limit, issues }) - return { id: request.id, ok: true, value: { candidates, issues } } + return { id: request.id, kind: 'result', value: { candidates, issues } } } - return { id: request.id, ok: true, value: await parseSession(request) } + return { id: request.id, kind: 'result', value: await parseSession(request) } } catch (err) { - return { id: request.id, ok: false, error: err instanceof Error ? err.message : String(err) } + return { + id: request.id, + kind: 'error', + error: err instanceof Error ? err.message : String(err) + } } } @@ -78,7 +82,7 @@ port.on('message', (request: OpenCodeSqliteParentMessage) => { // waiting out its timeout; fail that request fast instead. port.postMessage({ id: request.id, - ok: false, + kind: 'error', error: 'OpenCode SQLite worker result could not be serialized.' }) } diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-protocol.ts b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-protocol.ts index 0adda81a629..ce9226cb098 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite-worker-protocol.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite-worker-protocol.ts @@ -39,18 +39,27 @@ export type OpenCodeSqliteParseValue = { session: AiVaultSession | null } +// One acknowledged slice of a parse's index rows, sent before that parse's +// result. `batch` numbers them so a late ack cannot release the wrong one. +export type OpenCodeSqliteCaptureBatch = { + id: number + kind: 'batch' + batch: number + messages: SessionSearchCapturedMessage[] +} + +export type OpenCodeSqliteWorkerResult = { id: number; kind: 'result'; value: unknown } +export type OpenCodeSqliteWorkerFailure = { id: number; kind: 'error'; error: string } + +// Tagged rather than a boolean plus an optional field: the tag is what lets the +// client narrow to the batch shape without casting its payload. export type OpenCodeSqliteWorkerResponse = - | { id: number; ok: true; value: unknown; captureBatch?: number } - | { id: number; ok: false; error: string } + | OpenCodeSqliteCaptureBatch + | OpenCodeSqliteWorkerResult + | OpenCodeSqliteWorkerFailure export type OpenCodeSqliteCaptureAck = { id: number; kind: 'captureAck'; batch: number } export type OpenCodeSqliteParentMessage = OpenCodeSqliteWorkerRequest | OpenCodeSqliteCaptureAck -export type OpenCodeSqliteCaptureBatch = { - id: number - ok: true - captureBatch: number - value: SessionSearchCapturedMessage[] -} export type OpenCodeSqliteRequestBody = | Omit diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite.test.ts b/src/main/ai-vault/session-scanner-opencode-sqlite.test.ts index 2c4a075cc8c..bf46f429210 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite.test.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite.test.ts @@ -443,4 +443,47 @@ describe('parseOpenCodeSqliteSession', () => { expect(session!.firstUserPrompt).toBe('the real typed ask') }) + + it('still returns the session when a corrupt part blob breaks the preview read', async () => { + const { db, path } = createTempDb() + applyOpenCodeSqliteSchema(db) + insertOpenCodeSession(db, { + id: 'ses_badblob', + title: 'Ballast planning', + timeCreated: 1_777_634_000_000, + timeUpdated: 1_777_634_900_000 + }) + insertOpenCodeMessage(db, { + id: 'msg_1', + sessionId: 'ses_badblob', + role: 'user', + timeCreated: 1_777_634_000_000 + }) + insertOpenCodePart(db, { + id: 'part_ok', + messageId: 'msg_1', + sessionId: 'ses_badblob', + timeCreated: 10, + text: 'readable turn' + }) + // Truncated JSON, so the preview query's json_extract raises rather than + // returning NULL. One corrupt row must not cost the session its listing. + db.prepare( + `INSERT INTO part (id, message_id, session_id, time_created, time_updated, data) + VALUES ('part_bad', 'msg_1', 'ses_badblob', 20, 20, '{"type":')` + ).run() + db.close() + + const session = await parseOpenCodeSqliteSession({ + dbPath: path, + sessionId: 'ses_badblob', + platform: 'darwin' + }) + + expect(session).not.toBeNull() + expect(session!.sessionId).toBe('ses_badblob') + expect(session!.title).toBe('Ballast planning') + // The preview degrades to empty rather than taking the session with it. + expect(session!.previewMessages).toEqual([]) + }) }) diff --git a/src/main/ai-vault/session-scanner-opencode-sqlite.ts b/src/main/ai-vault/session-scanner-opencode-sqlite.ts index b9e869fb058..1ff0d6f07fc 100644 --- a/src/main/ai-vault/session-scanner-opencode-sqlite.ts +++ b/src/main/ai-vault/session-scanner-opencode-sqlite.ts @@ -12,6 +12,10 @@ import { shouldCaptureFullFirstUserPrompt } from './session-scanner-first-user-prompt' import { readOpenCodeDatabaseAsync } from './session-scanner-opencode-sqlite-open' +import { + canCountOpenCodeMessages, + canReadOpenCodeMessageParts +} from './session-scanner-opencode-sqlite-schema' import { normalizeTitleText } from './session-scanner-values' import type SyncDatabase from '../sqlite/sync-database' import { columnExists, tableExists } from '../opencode-usage/schema-helpers' @@ -72,14 +76,6 @@ function sessionNumberColumnSelect(db: SyncDatabase, columnName: string): string return columnExists(db, 'session', columnName) ? `s.${columnName}` : '0' } -function canCountOpenCodeMessages(db: SyncDatabase): boolean { - return ( - tableExists(db, 'message') && - columnExists(db, 'message', 'session_id') && - columnExists(db, 'message', 'data') - ) -} - function buildSessionQuery(db: SyncDatabase): string { const messageCountSubquery = canCountOpenCodeMessages(db) ? `(SELECT COUNT(*) FROM message m @@ -156,14 +152,7 @@ function extractPartText(partData: string): string | null { } function readFirstUserPromptFromOpenCodeDb(db: SyncDatabase, sessionId: string): string | null { - if ( - !canCountOpenCodeMessages(db) || - !tableExists(db, 'part') || - !columnExists(db, 'message', 'id') || - !columnExists(db, 'part', 'message_id') || - !columnExists(db, 'part', 'time_created') || - !columnExists(db, 'part', 'data') - ) { + if (!canReadOpenCodeMessageParts(db)) { return null } @@ -208,14 +197,7 @@ function readFirstUserPromptFromOpenCodeDb(db: SyncDatabase, sessionId: string): } function buildPreviewQuery(db: SyncDatabase): string | null { - if ( - !canCountOpenCodeMessages(db) || - !tableExists(db, 'part') || - !columnExists(db, 'message', 'id') || - !columnExists(db, 'part', 'message_id') || - !columnExists(db, 'part', 'time_created') || - !columnExists(db, 'part', 'data') - ) { + if (!canReadOpenCodeMessageParts(db)) { return null } return `SELECT json_extract(m.data, '$.role') AS role, @@ -234,6 +216,29 @@ function buildPreviewQuery(db: SyncDatabase): string | null { LIMIT ?` } +/** + * Run the preview query, or null when it cannot be read. + * + * `json_extract` raises on a malformed part blob, so an unguarded read here + * would drop the whole session from the list over one corrupt row. Degrade to + * "no preview" instead, matching every other read in this module. + */ +function readPreviewRows( + db: SyncDatabase, + previewSql: string, + sessionId: string +): PreviewRow[] | null { + try { + return db.prepare(previewSql).all(sessionId, OPENCODE_SQLITE_PREVIEW_LIMIT + 1) as PreviewRow[] + } catch (error) { + console.warn( + '[ai-vault] opencode preview skipped', + error instanceof Error ? error.name : 'ReadError' + ) + return null + } +} + /** * Parse a single OpenCode session from the SQLite database into an * `AiVaultSession`. Reads session metadata (title, cwd, model, tokens, cost) @@ -298,14 +303,11 @@ async function readSession(args: { updateTimeline(accumulator, row.time_updated) const previewSql = buildPreviewQuery(db) - if (previewSql) { - await captureOpenCodeSession(db, sessionId) - // Why: SQL already dropped anything older than the newest-N window, so the - // accumulator never shifts and cannot detect the truncation itself. Ask for - // one extra row so an exactly-full window is not mistaken for a trimmed one. - const probedRows = db - .prepare(previewSql) - .all(sessionId, OPENCODE_SQLITE_PREVIEW_LIMIT + 1) as PreviewRow[] + // Why: SQL already dropped anything older than the newest-N window, so the + // accumulator never shifts and cannot detect the truncation itself. Ask for + // one extra row so an exactly-full window is not mistaken for a trimmed one. + const probedRows = previewSql ? readPreviewRows(db, previewSql, sessionId) : null + if (probedRows) { if (probedRows.length > OPENCODE_SQLITE_PREVIEW_LIMIT) { accumulator.previewMessagesTruncated = true } @@ -340,6 +342,8 @@ async function readSession(args: { } } + await captureOpenCodeSession(db, sessionId) + // Why: list preview only joins the newest messages. On-demand copy needs the // session's earliest real user text part, not a later turn still in the window. if (shouldCaptureFullFirstUserPrompt()) { diff --git a/src/main/ai-vault/session-scanner-opencode-worker-host.ts b/src/main/ai-vault/session-scanner-opencode-worker-host.ts new file mode 100644 index 00000000000..e9667c29bc7 --- /dev/null +++ b/src/main/ai-vault/session-scanner-opencode-worker-host.ts @@ -0,0 +1,108 @@ +import type { Worker } from 'node:worker_threads' +import type { OpenCodeSqliteWorkerResponse } from './session-scanner-opencode-sqlite-worker-protocol' +import { errorMessage } from './session-scanner-values' + +export type WorkerFactory = () => Worker + +export const IDLE_TEARDOWN_MS = 30_000 + +/** + * Owns the lifetime of the one OpenCode SQLite worker thread: lazy spawn, + * listener wiring, teardown, and idle expiry. It holds no request state, so + * every decision about which call a message belongs to stays with the client. + */ +export class OpenCodeSqliteWorkerHost { + private worker: Worker | null = null + private idleTimer: NodeJS.Timeout | null = null + private cleanupListeners: (() => void) | null = null + private loggedUnavailable = false + + constructor( + private readonly options: { + factory: WorkerFactory + log: (message: string) => void + onMessage: (response: OpenCodeSqliteWorkerResponse) => void + onError: (error: Error) => void + onExit: (code: number) => void + /** Nothing active and nothing queued, checked again when the idle timer fires. */ + isIdle: () => boolean + } + ) {} + + get current(): Worker | null { + return this.worker + } + + /** The live worker, spawning one if needed; null when no worker can be had. */ + ensure(): Worker | null { + if (this.worker) { + return this.worker + } + try { + const worker = this.options.factory() + const onMessage = (response: OpenCodeSqliteWorkerResponse): void => + this.options.onMessage(response) + const onError = (error: Error): void => this.options.onError(error) + const onExit = (code: number): void => this.options.onExit(code) + worker.on('message', onMessage) + worker.on('error', onError) + worker.on('exit', onExit) + this.cleanupListeners = () => { + worker.off('message', onMessage) + worker.off('error', onError) + worker.off('exit', onExit) + } + // Never keep the app alive for a scan worker. + worker.unref?.() + this.worker = worker + return worker + } catch (err) { + // Why (#8864): never fall back to synchronous SQLite reads here; a missing + // bundle or resource-exhausted spawn must omit OpenCode history rather than + // reintroduce the main-process hang this worker boundary prevents. + if (!this.loggedUnavailable) { + this.loggedUnavailable = true + this.options.log( + `OpenCode SQLite worker unavailable; skipping its history. ${errorMessage(err)}` + ) + } + return null + } + } + + destroy(): void { + this.clearIdleTimer() + const worker = this.worker + this.worker = null + if (!worker) { + return + } + this.cleanupListeners?.() + this.cleanupListeners = null + worker.removeAllListeners() + void worker.terminate().catch(() => undefined) + } + + scheduleIdleTeardown(): void { + this.clearIdleTimer() + if (!this.worker) { + return + } + this.idleTimer = setTimeout(() => { + this.idleTimer = null + // Re-checked here: a request arriving as the timer fires must never be + // lost to a self-exiting worker. + if (this.options.isIdle()) { + this.destroy() + } + }, IDLE_TEARDOWN_MS) + this.idleTimer.unref?.() + } + + clearIdleTimer(): void { + if (this.idleTimer) { + clearTimeout(this.idleTimer) + this.idleTimer = null + } + } +} diff --git a/src/main/ai-vault/session-scanner-parse-cache.ts b/src/main/ai-vault/session-scanner-parse-cache.ts index c86f7d75bc8..1273d9677dc 100644 --- a/src/main/ai-vault/session-scanner-parse-cache.ts +++ b/src/main/ai-vault/session-scanner-parse-cache.ts @@ -177,8 +177,13 @@ async function parseCachedInLane( // Codex titles come from session_index.jsonl, which mtime+size can't see. // Remote counterpart: remote-session-scanner.ts's reusedCodexTitleRefresh. if (entry.session && candidate.agent === 'codex') { + // Why: the refresh returns the same reference when nothing changed, and a + // list scan runs every ~5 s; writing regardless is one UPDATE per session. + const previous = entry.session entry.session = await refreshCachedCodexMetadata(candidate, entry.session) - sink?.updateMetadata?.(candidate, entry.session) + if (entry.session !== previous) { + sink?.updateMetadata?.(candidate, entry.session) + } } storeSessionParseCacheEntry(file.path, entry) return entry.session @@ -318,6 +323,9 @@ async function parseResumableCandidate(args: { } return { value: entry, + // Why: the index must see the state without the partial line. finalize only + // reads the accumulator (rows are emitted in consumeLine), so this second + // call recomputes metadata without re-emitting any captured message. session: displayState === state ? session : await state.finalize(args.platform), byteOffset: readResult.consumedThrough } diff --git a/src/main/ai-vault/session-scanner-service-entry-search-init.test.ts b/src/main/ai-vault/session-scanner-service-entry-search-init.test.ts new file mode 100644 index 00000000000..f0ca69f00c4 --- /dev/null +++ b/src/main/ai-vault/session-scanner-service-entry-search-init.test.ts @@ -0,0 +1,53 @@ +import { beforeAll, expect, it, vi } from 'vitest' +import { AI_VAULT_SERVICE_PROTOCOL_VERSION } from './session-scanner-service-protocol' + +// The service entry's stdio is piped to the parent's console +// (session-scanner-service-spawn.ts), so anything it logs leaves the child. + +const INDEX_PATH = '/Users/somebody/Library/orca/session-index.sqlite' + +vi.mock('../ai-vault-search/session-search-service', () => ({ + SessionSearchService: class { + constructor() { + throw new Error(`unable to open database file ${INDEX_PATH}`) + } + } +})) +vi.mock('./session-scanner', () => ({ scanAiVaultSessions: vi.fn() })) +vi.mock('./session-parse-cache-persistence', () => ({ + flushSessionParseCachePersist: vi.fn(() => Promise.resolve()), + initSessionParseCachePersistence: vi.fn() +})) +vi.mock('./session-subagent-reader', () => ({ + listLocalAiVaultSubagentSessions: vi.fn(() => Promise.resolve({ sessions: [], issues: [] })) +})) + +let logged: unknown[] = [] + +beforeAll(async () => { + process.send = (() => true) as typeof process.send + vi.spyOn(console, 'error').mockImplementation((...args: unknown[]) => { + logged = args + }) + await import('./session-scanner-service-entry') + process.emit( + 'message', + { + type: 'init', + protocol: AI_VAULT_SERVICE_PROTOCOL_VERSION, + sessionSearch: { databasePath: INDEX_PATH, enabled: true, historyDays: null } + } as never, + undefined as never + ) +}) + +it('logs only the error name when the search index cannot be opened', () => { + expect(logged[0]).toBe('[ai-vault] session search index unavailable:') + expect(logged[1]).toBe('Error') + + // Neither the thrown object nor the user's index path may reach the pipe. + for (const value of logged) { + expect(value).not.toBeInstanceOf(Error) + expect(String(value)).not.toContain(INDEX_PATH) + } +}) diff --git a/src/main/ai-vault/session-scanner-service-entry.ts b/src/main/ai-vault/session-scanner-service-entry.ts index c9d3ebdaca5..0955e4cb6e5 100644 --- a/src/main/ai-vault/session-scanner-service-entry.ts +++ b/src/main/ai-vault/session-scanner-service-entry.ts @@ -186,7 +186,7 @@ async function shutdown(): Promise { } await Promise.allSettled([cacheLane, interactiveLane]) await flushSessionParseCachePersist() - sessionSearch?.dispose() + await sessionSearch?.close() process.disconnect?.() } @@ -204,7 +204,12 @@ process.on('message', (raw: AiVaultServiceParentMessage) => { try { sessionSearch = new SessionSearchService(raw.sessionSearch) } catch (error) { - console.error('[ai-vault] session search index unavailable:', error) + // Name only: this stream is piped to the parent's console, so nothing + // from a transcript-bearing failure may ride out on it. + console.error( + '[ai-vault] session search index unavailable:', + error instanceof Error ? error.name : 'IndexOpenError' + ) } } send({ type: 'ready', protocol: AI_VAULT_SERVICE_PROTOCOL_VERSION, pid: process.pid }) diff --git a/src/main/ai-vault/session-scanner-service-env.ts b/src/main/ai-vault/session-scanner-service-env.ts index ba06a35a358..a29dd9d2108 100644 --- a/src/main/ai-vault/session-scanner-service-env.ts +++ b/src/main/ai-vault/session-scanner-service-env.ts @@ -10,7 +10,6 @@ // What Node and libuv need to start and resolve a home, temp dir and locale. // Exported for sibling plain-node forks (the WSL transcript fs process). export const RUNTIME_ENV_ALLOWLIST = [ - 'ORCA_BACKGROUND_LAUNCH', 'PATH', 'HOME', 'USERPROFILE', diff --git a/src/main/ai-vault/session-search-capture.ts b/src/main/ai-vault/session-search-capture.ts index aa922fbadeb..586fa9eca18 100644 --- a/src/main/ai-vault/session-search-capture.ts +++ b/src/main/ai-vault/session-search-capture.ts @@ -32,12 +32,10 @@ export type SessionSearchIndexUpdate = { export type SessionSearchIndexResult = Pick -/** Streaming writes receive final metadata only when parsing completes. */ -export type SessionSearchIndexWrite = - | SessionSearchIndexUpdate - | (Omit & { - result: Promise - }) +/** Final metadata and cursor arrive only when the streamed parse completes. */ +export type SessionSearchIndexWrite = Omit & { + result: Promise +} export type SessionSearchFileIdentity = { dev: number; ino: number } | null @@ -48,7 +46,6 @@ export type SessionSearchIndexedFile = { } export type SessionSearchIndexSink = { - streamingCapture?: boolean acceptsCandidate?(candidate: SessionFileCandidate): boolean updateMetadata?(candidate: SessionFileCandidate, session: AiVaultSession): void /** @@ -117,14 +114,6 @@ export function withoutSessionSearchCapture(fn: () => T): T { return captureStorage.run(null, fn) } -export async function withSessionSearchCapture( - fn: () => Promise -): Promise<{ value: T; messages: SessionSearchCapturedMessage[] }> { - const scope = { messages: [] as SessionSearchCapturedMessage[] } - const value = await captureStorage.run(scope, fn) - return { value, messages: scope.messages } -} - export async function checkpointSessionSearchCapture(): Promise { const signal = getSessionSearchCaptureSignal() throwIfAiVaultScanCancelled(signal) diff --git a/src/main/ai-vault/session-search-indexed-parse.ts b/src/main/ai-vault/session-search-indexed-parse.ts index 839c4891653..9a36dc82ac1 100644 --- a/src/main/ai-vault/session-search-indexed-parse.ts +++ b/src/main/ai-vault/session-search-indexed-parse.ts @@ -3,7 +3,6 @@ import type { AiVaultSession } from '../../shared/ai-vault-types' import type { SessionSearchIndexSink, SessionSearchIndexUpdate } from './session-search-capture' import { getSessionSearchCaptureSignal, - withSessionSearchCapture, withStreamingSessionSearchCapture } from './session-search-capture' import { SessionSearchMessageChannel } from './session-search-message-channel' @@ -23,17 +22,6 @@ export async function captureIndexedSessionParse( throwIfAiVaultScanCancelled(signal) return result } - if (!sink.streamingCapture) { - const captured = await withSessionSearchCapture(read) - await sink.apply({ - ...base, - signal, - session: captured.value.session, - byteOffset: captured.value.byteOffset, - messages: captured.messages - }) - return captured.value.value - } const channel = new SessionSearchMessageChannel() const stop = (): void => channel.stop() signal?.addEventListener('abort', stop, { once: true }) diff --git a/src/main/ai-vault/session-search-message-channel.test.ts b/src/main/ai-vault/session-search-message-channel.test.ts new file mode 100644 index 00000000000..c87293bb08e --- /dev/null +++ b/src/main/ai-vault/session-search-message-channel.test.ts @@ -0,0 +1,40 @@ +import { expect, it } from 'vitest' +import { SessionSearchMessageChannel } from './session-search-message-channel' + +const message = (text: string) => ({ role: 'user' as const, text, timestamp: null }) + +/** Resolves to 'stalled' when a producer is never woken, instead of hanging the run. */ +function settledOrStalled(promises: Promise[]): Promise { + return Promise.race([ + Promise.all(promises).then(() => 'settled'), + new Promise((resolve) => setTimeout(() => resolve('stalled'), 250)) + ]) +} + +it('resumes every producer waiting on a checkpoint, not just the last one', async () => { + const channel = new SessionSearchMessageChannel() + channel.push(message('first')) + const first = channel.checkpoint() + channel.push(message('second')) + const second = channel.checkpoint() + const drained: string[] = [] + const consumer = (async () => { + for await (const value of channel) { + drained.push(value.text) + } + })() + channel.close() + await consumer + expect(drained).toEqual(['first', 'second']) + expect(await settledOrStalled([first, second])).toBe('settled') +}) + +it('releases checkpoint waiters when the consumer stops', async () => { + const channel = new SessionSearchMessageChannel() + channel.push(message('first')) + const first = channel.checkpoint() + channel.push(message('second')) + const second = channel.checkpoint() + channel.stop() + expect(await settledOrStalled([first, second])).toBe('settled') +}) diff --git a/src/main/ai-vault/session-search-message-channel.ts b/src/main/ai-vault/session-search-message-channel.ts index 74cd28f29b4..3b12d007188 100644 --- a/src/main/ai-vault/session-search-message-channel.ts +++ b/src/main/ai-vault/session-search-message-channel.ts @@ -4,7 +4,7 @@ import type { SessionSearchCapturedMessage } from './session-search-capture' export class SessionSearchMessageChannel implements AsyncIterable { private queued: SessionSearchCapturedMessage[] = [] private wake: (() => void) | null = null - private drained: (() => void) | null = null + private drained: (() => void)[] = [] private ended = false private stopped = false private failure: unknown @@ -19,8 +19,10 @@ export class SessionSearchMessageChannel implements AsyncIterable { - this.drained = resolve + this.drained.push(resolve) }) } close(error?: unknown): void { @@ -31,9 +33,17 @@ export class SessionSearchMessageChannel implements AsyncIterable { while (!this.stopped) { const batch = this.queued @@ -41,8 +51,7 @@ export class SessionSearchMessageChannel implements AsyncIterable client.parse(args), controller.signal) - worker.emit('message', { id: worker.requests[0].id, ok: true, value: { session: null } }) + worker.emit('message', { id: worker.requests[0].id, kind: 'result', value: { session: null } }) expect(await parsed).toBeNull() expect(getEventListeners(controller.signal, 'abort')).toHaveLength(0) controller.abort() @@ -126,7 +126,6 @@ it('unblocks a local capture checkpoint on abort while an ordinary list still co previousByteOffset: 0 } const sink = { - streamingCapture: true, indexedFile: () => null, markStale() {}, async apply() { diff --git a/src/main/ai-vault/session-search-opencode-capture-channel.ts b/src/main/ai-vault/session-search-opencode-capture-channel.ts new file mode 100644 index 00000000000..fcb38b605b6 --- /dev/null +++ b/src/main/ai-vault/session-search-opencode-capture-channel.ts @@ -0,0 +1,75 @@ +import type { Worker } from 'node:worker_threads' +import type { OpenCodeSqliteCaptureBatch } from './session-scanner-opencode-sqlite-worker-protocol' +import { AsyncResource } from 'node:async_hooks' +import { + captureSessionSearchMessage, + checkpointSessionSearchCapture, + type SessionSearchCapturedMessage +} from './session-search-capture' + +// The main-thread end of the worker's capture batch/ack loop: one batch in +// flight, acknowledged only once the caller's sink has taken it. + +export type OpenCodeCaptureConsumer = (messages: SessionSearchCapturedMessage[]) => Promise + +/** Binds the current capture scope: AsyncLocalStorage does not survive the worker hop. */ +export function bindOpenCodeCaptureConsumer(): OpenCodeCaptureConsumer { + return AsyncResource.bind(async (messages: SessionSearchCapturedMessage[]) => { + for (const message of messages) { + captureSessionSearchMessage(message) + } + await checkpointSessionSearchCapture() + }) +} + +type DeadlinedCall = { + capture?: OpenCodeCaptureConsumer + timer: NodeJS.Timeout | null + timeoutMs: number +} + +// Reset rather than cleared: total production time stays unbounded (that is the +// point of the credit loop), but each individual stall is still capped, so a +// backlogged index writer costs one scan issue instead of wedging the client. +function restartDeadline(call: DeadlinedCall, onTimeout: () => void): void { + if (call.timer) { + clearTimeout(call.timer) + } + call.timer = setTimeout(onTimeout, call.timeoutMs) + call.timer.unref?.() +} + +export function receiveOpenCodeCaptureBatch(args: { + call: DeadlinedCall + batch: OpenCodeSqliteCaptureBatch + worker: Worker | null + isActive: () => boolean + onTimeout: () => void + onProtocolViolation: (error: Error) => void + onConsumerError: (error: Error) => void +}): void { + const { call } = args + if (!call.capture) { + args.onProtocolViolation(new Error('Unexpected OpenCode capture batch.')) + return + } + restartDeadline(call, args.onTimeout) + void call + .capture(args.batch.messages) + .then(() => { + if (!args.isActive()) { + return + } + restartDeadline(call, args.onTimeout) + args.worker?.postMessage({ + id: args.batch.id, + kind: 'captureAck', + batch: args.batch.batch + }) + }) + .catch((error) => { + if (args.isActive()) { + args.onConsumerError(error instanceof Error ? error : new Error(String(error))) + } + }) +} diff --git a/src/main/ai-vault/session-search-opencode-content.ts b/src/main/ai-vault/session-search-opencode-content.ts index 8ba6be61736..684eedfe18a 100644 --- a/src/main/ai-vault/session-search-opencode-content.ts +++ b/src/main/ai-vault/session-search-opencode-content.ts @@ -1,33 +1,76 @@ import type SyncDatabase from '../sqlite/sync-database' import { captureIndexableText, toolCallText } from './session-search-content' +import { canReadOpenCodeMessageParts } from './session-scanner-opencode-sqlite-schema' import { checkpointSessionSearchCapture, isSessionSearchCaptureActive } from './session-search-capture' +const SESSION_PARTS_SQL = `SELECT json_extract(m.data, '$.role') AS role, + p.data AS data, p.time_created AS ts FROM message m JOIN part p ON p.message_id = m.id + WHERE m.session_id = ? ORDER BY m.time_created, m.id, p.time_created, p.id` + +type OpenCodePartRow = { + role: unknown + part: { + type?: string + text?: string + tool?: string + state?: { input?: unknown; output?: string } + } + ts: unknown +} + +function decodePart(data: unknown): OpenCodePartRow['part'] | null { + try { + const parsed = JSON.parse(String(data)) as unknown + return parsed && typeof parsed === 'object' && !Array.isArray(parsed) + ? (parsed as OpenCodePartRow['part']) + : null + } catch { + return null + } +} + +/** + * Yield every decodable part of one session. + * + * A read failure ends the stream instead of throwing: search coverage degrades + * to no rows for this session, which must never cost the session its place in + * the list. Consumer-thrown cancellation resumes the generator with a `return` + * completion, so it never reaches the catch and still propagates to the caller. + */ +function* readOpenCodeSessionParts( + db: SyncDatabase, + sessionId: string +): Generator { + try { + for (const row of db.prepare(SESSION_PARTS_SQL).iterate(sessionId)) { + const part = decodePart(row.data) + if (part) { + yield { role: row.role, part, ts: row.ts } + } + } + } catch (error) { + console.warn( + '[ai-vault] opencode search capture skipped', + error instanceof Error ? error.name : 'ReadError' + ) + } +} + /** The preview ring is deliberately small; search consumes every part once. */ export async function captureOpenCodeSession(db: SyncDatabase, sessionId: string): Promise { - if (!isSessionSearchCaptureActive()) { + if (!isSessionSearchCaptureActive() || !canReadOpenCodeMessageParts(db)) { return } - const rows = db - .prepare(`SELECT json_extract(m.data, '$.role') AS role, - p.data AS data, p.time_created AS ts FROM message m JOIN part p ON p.message_id = m.id - WHERE m.session_id = ? ORDER BY m.time_created, m.id, p.time_created, p.id`) - .iterate(sessionId) - for (const row of rows) { - const part = JSON.parse(String(row.data)) as { - type?: string - text?: string - tool?: string - state?: { input?: unknown; output?: string } - } + for (const { role, part, ts } of readOpenCodeSessionParts(db, sessionId)) { if (part.type === 'text' && typeof part.text === 'string') { - captureIndexableText(row.role === 'user' ? 'user' : 'assistant', part.text, row.ts) + captureIndexableText(role === 'user' ? 'user' : 'assistant', part.text, ts) } else if (part.type === 'tool') { - captureIndexableText('tool', toolCallText(part.tool, part.state?.input), row.ts) + captureIndexableText('tool', toolCallText(part.tool, part.state?.input), ts) if (typeof part.state?.output === 'string') { - captureIndexableText('tool', part.state.output, row.ts) + captureIndexableText('tool', part.state.output, ts) } } await checkpointSessionSearchCapture() diff --git a/src/main/ai-vault/session-search-opencode-worker-capture.ts b/src/main/ai-vault/session-search-opencode-worker-capture.ts index cbbee794b81..f04c293d3cd 100644 --- a/src/main/ai-vault/session-search-opencode-worker-capture.ts +++ b/src/main/ai-vault/session-search-opencode-worker-capture.ts @@ -39,7 +39,7 @@ export class OpenCodeWorkerSearchCapture { try { await new Promise((resolve) => { this.acknowledgeBatch = resolve - this.send({ id: this.id, ok: true, captureBatch: sequence, value: messages }) + this.send({ id: this.id, kind: 'batch', batch: sequence, messages }) }) } finally { this.acknowledgeBatch = null diff --git a/src/main/ai-vault/session-search-opencode-worker-receiver.ts b/src/main/ai-vault/session-search-opencode-worker-receiver.ts deleted file mode 100644 index 3fa7df3b7b4..00000000000 --- a/src/main/ai-vault/session-search-opencode-worker-receiver.ts +++ /dev/null @@ -1,94 +0,0 @@ -import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation' -import type { Worker } from 'node:worker_threads' -import type { - OpenCodeSqliteWorkerRequest, - OpenCodeSqliteWorkerResponse -} from './session-scanner-opencode-sqlite-worker-protocol' -import { AsyncResource } from 'node:async_hooks' -import { - captureSessionSearchMessage, - checkpointSessionSearchCapture, - getSessionSearchCaptureSignal, - type SessionSearchCapturedMessage -} from './session-search-capture' - -export type OpenCodeCaptureConsumer = (messages: SessionSearchCapturedMessage[]) => Promise - -export function bindOpenCodeCaptureConsumer(): OpenCodeCaptureConsumer { - return AsyncResource.bind(async (messages: SessionSearchCapturedMessage[]) => { - for (const message of messages) { - captureSessionSearchMessage(message) - } - await checkpointSessionSearchCapture() - }) -} - -export function receiveOpenCodeCaptureBatch(args: { - call: { capture?: OpenCodeCaptureConsumer; timer: NodeJS.Timeout | null; timeoutMs: number } - response: Extract - worker: Worker | null - isActive: () => boolean - onTimeout: () => void - onError: (error: Error) => void -}): void { - const { call } = args - if (!call.capture) { - args.onError(new Error('Unexpected OpenCode capture batch.')) - return - } - // Backpressure belongs to the writer; the worker deadline covers time spent producing. - if (call.timer) { - clearTimeout(call.timer) - call.timer = null - } - void call - .capture(args.response.value as SessionSearchCapturedMessage[]) - .then(() => { - if (!args.isActive()) { - return - } - call.timer = setTimeout(args.onTimeout, call.timeoutMs) - call.timer.unref?.() - args.worker?.postMessage({ - id: args.response.id, - kind: 'captureAck', - batch: args.response.captureBatch - }) - }) - .catch((error) => { - if (args.isActive()) { - args.onError(error instanceof Error ? error : new Error(String(error))) - } - }) -} - -/** The request owns its abort listener until either queued or active work settles. */ -export function bindOpenCodeCaptureCancellation( - resolve: (value: unknown) => void, - reject: (error: Error) => void, - cancel: () => void -): { resolve: typeof resolve; reject: typeof reject } { - const signal = getSessionSearchCaptureSignal() - throwIfAiVaultScanCancelled(signal) - signal?.addEventListener('abort', cancel, { once: true }) - const cleanup = (): void => signal?.removeEventListener('abort', cancel) - return { - resolve: (value) => { - cleanup() - resolve(value) - }, - reject: (error) => { - cleanup() - reject(error) - } - } -} - -export type OpenCodePendingCall = { - request: OpenCodeSqliteWorkerRequest - timeoutMs: number - resolve: (value: unknown) => void - reject: (error: Error) => void - timer: NodeJS.Timeout | null - capture?: OpenCodeCaptureConsumer -} diff --git a/src/main/ipc/ai-vault.ts b/src/main/ipc/ai-vault.ts index 98cdb579a98..56ad64d7cd3 100644 --- a/src/main/ipc/ai-vault.ts +++ b/src/main/ipc/ai-vault.ts @@ -19,6 +19,7 @@ import { mergeAiVaultListResults } from '../ai-vault/session-list-results' import type { AiVaultSearchArgs } from '../../shared/ai-vault-search-types' +import { projectSessionSearchResult } from '../../shared/ai-vault-search-projection' import { scanSshAiVaultSessions } from '../ai-vault/ssh-session-list' import { AiVaultScanCoordinator } from '../ai-vault/ai-vault-scan-coordinator' import type { AiVaultDeleteSessionArgs } from '../../shared/ai-vault-session-deletion' @@ -260,9 +261,10 @@ export function registerAiVaultHandlers(options: AiVaultHandlerOptions = {}): vo } }) // Local-only: the search index is built beside the transcripts on this host, - // so a remote scope has nothing to consult here. + // so a remote scope has nothing to consult here. Projected all the same, so the + // renderer sees one result shape whether the host is local or remote. ipcMain.handle('aiVault:searchSessions', (_event, args: AiVaultSearchArgs) => - searchAiVaultSessions(args) + searchAiVaultSessions(args).then(projectSessionSearchResult) ) ipcMain.handle('aiVault:searchCoverage', () => readAiVaultSearchCoverage()) ipcMain.handle('aiVault:searchIndexSize', () => ({ diff --git a/src/main/ipc/settings.ts b/src/main/ipc/settings.ts index 55d9df904df..501e0b59861 100644 --- a/src/main/ipc/settings.ts +++ b/src/main/ipc/settings.ts @@ -37,7 +37,7 @@ import { normalizeComputerAwakeMode } from '../../shared/computer-awake-mode' import { - applyAiVaultSearchSettings, + applyAiVaultSearchSettingsChange, installAiVaultSearchSettingsSource } from '../ai-vault-search/session-search-enablement' import { resolveAiVaultSearchSettings } from '../../shared/ai-vault-search-settings' @@ -278,9 +278,9 @@ export function registerSettingsHandlers( applyAppIcon(result.appIcon) } if ('aiVaultSearch' in sanitizedArgs) { - await applyAiVaultSearchSettings(result, { - persist: () => store.flushPendingOrThrowAsync({ drainToStableGeneration: false }) - }) + applyAiVaultSearchSettingsChange(before, result, () => + store.flushPendingOrThrowAsync({ drainToStableGeneration: false }) + ) } // Why: telemetry-plan.md§Settings — fire `settings_changed` only for diff --git a/src/main/runtime/rpc/methods/ai-vault-search.test.ts b/src/main/runtime/rpc/methods/ai-vault-search.test.ts index cda17055345..5a12274578c 100644 --- a/src/main/runtime/rpc/methods/ai-vault-search.test.ts +++ b/src/main/runtime/rpc/methods/ai-vault-search.test.ts @@ -269,14 +269,44 @@ describe('aiVault.searchCoverage handler', () => { ).resolves.toMatchObject({ ok: true, result: COVERAGE }) expect(readAiVaultSearchCoverage).toHaveBeenCalledWith(controller.signal) }) +}) - it('rejects a non-runtime execution host id', async () => { +describe('host-local execution boundary', () => { + // Every one of these runs on this host's own index. An id naming a host this + // process does not execute on must be refused, not answered locally. + const methods = [ + ['aiVault.searchSessions', { query: 'q' }], + ['aiVault.searchCoverage', {}], + ['aiVault.searchIndexStatus', {}], + ['aiVault.configureSessionSearch', { enabled: true }] + ] as const + + it.each(['ssh:build-server', 'local', 'not-a-host'])( + 'refuses %s on every search method instead of answering with this host', + async (executionHostId) => { + const dispatcher = makeDispatcher() + for (const [method, params] of methods) { + await expect( + dispatcher.dispatch(makeRequest(method, { ...params, executionHostId })) + ).resolves.toMatchObject({ ok: false }) + } + expect(searchAiVaultSessions).not.toHaveBeenCalled() + expect(readAiVaultSearchCoverage).not.toHaveBeenCalled() + expect(readAiVaultSearchIndexStatus).not.toHaveBeenCalled() + expect(configureAiVaultSessionSearch).not.toHaveBeenCalled() + } + ) + + it('accepts a runtime id on every search method without letting it route the call', async () => { const dispatcher = makeDispatcher() - - await expect( - dispatcher.dispatch(makeRequest('aiVault.searchCoverage', { executionHostId: 'ssh:box' })) - ).resolves.toMatchObject({ ok: false }) - expect(readAiVaultSearchCoverage).not.toHaveBeenCalled() + for (const [method, params] of methods) { + await expect( + dispatcher.dispatch(makeRequest(method, { ...params, executionHostId: 'runtime:env-1' })) + ).resolves.toMatchObject({ ok: true }) + } + expect(searchAiVaultSessions.mock.calls[0]?.[0]).not.toHaveProperty('executionHostId') + expect(configureAiVaultSessionSearch.mock.calls[0]?.[0]).not.toHaveProperty('executionHostId') + expect(sshSearchAiVault).not.toHaveBeenCalled() }) }) diff --git a/src/main/runtime/rpc/methods/ai-vault.ts b/src/main/runtime/rpc/methods/ai-vault.ts index 1932502a918..7efbf5d431b 100644 --- a/src/main/runtime/rpc/methods/ai-vault.ts +++ b/src/main/runtime/rpc/methods/ai-vault.ts @@ -11,6 +11,7 @@ import { AI_VAULT_SESSION_TITLE_REQUEST_MAX_COUNT } from '../../../../shared/ai- import type { AiVaultPrepareSessionResumeArgs } from '../../../../shared/ai-vault-resume-preparation' import { LOCAL_EXECUTION_HOST_ID, parseExecutionHostId } from '../../../../shared/execution-host' import { describeAiVaultScanError } from '../../../../shared/ai-vault-scan-error-message' +import { SESSION_SEARCH_METHODS } from '../../../../shared/ai-vault-search-rpc-methods' import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { assertLegacyAiVaultResumeAllowed, @@ -23,6 +24,13 @@ import { const AI_VAULT_SCOPE_PATH_MAX_LENGTH = 4096 const AI_VAULT_LIMIT_MAX = 2000 +// Why: this is the whole SSH/foreign-host boundary for the aiVault surface. The +// scan and the index are host-local, so a caller must not be able to name a host +// this process does not execute on — an `ssh:` or `local` id is refused here +// rather than silently answered with this runtime's own transcripts +// (docs/reference/ssh-execution-boundary.md rule 1). A `runtime:` id names the +// *client's* saved environment, whose id this host never learns, so it is +// accepted for restamping only and never routes anything. const executionHostIdSchema = z.string().transform((value, ctx): `runtime:${string}` => { const parsed = parseExecutionHostId(value) if (parsed?.kind === 'runtime') { @@ -102,25 +110,25 @@ export const AiVaultConfigureSessionSearchParams = SessionSearchConfigureSchema. export const AI_VAULT_METHODS: RpcMethod[] = [ defineMethod({ - name: 'aiVault.sshSearchSessions', + name: SESSION_SEARCH_METHODS.query.runtimeSsh, params: SessionSearchQuerySchema.extend({ targetId: z.string().min(1).max(512) }), handler: ({ targetId, ...params }, { runtime, signal }) => runtime.sshSearchAiVault(targetId, 'query', params, signal) }), defineMethod({ - name: 'aiVault.sshSearchIndexStatus', + name: SESSION_SEARCH_METHODS.status.runtimeSsh, params: z.object({ targetId: z.string().min(1).max(512) }), handler: ({ targetId }, { runtime, signal }) => runtime.sshSearchAiVault(targetId, 'status', {}, signal) }), defineMethod({ - name: 'aiVault.sshSearchConfigure', + name: SESSION_SEARCH_METHODS.configure.runtimeSsh, params: SessionSearchConfigureSchema.extend({ targetId: z.string().min(1).max(512) }), handler: ({ targetId, ...params }, { runtime, signal }) => runtime.sshSearchAiVault(targetId, 'configure', params, signal) }), defineMethod({ - name: 'aiVault.searchSessions', + name: SESSION_SEARCH_METHODS.query.runtime, params: AiVaultSearchSessionsParams, // Why: the index lives with the transcripts, so this runs on the host the // client addressed; the id only names that host, it never redirects the search. @@ -133,12 +141,12 @@ export const AI_VAULT_METHODS: RpcMethod[] = [ handler: (_params, { runtime, signal }) => runtime.readAiVaultSearchCoverage(signal) }), defineMethod({ - name: 'aiVault.searchIndexStatus', + name: SESSION_SEARCH_METHODS.status.runtime, params: z.object({ executionHostId: executionHostIdSchema.optional() }), handler: (_params, { runtime }) => runtime.readAiVaultSearchIndexStatus() }), defineMethod({ - name: 'aiVault.configureSessionSearch', + name: SESSION_SEARCH_METHODS.configure.runtime, params: AiVaultConfigureSessionSearchParams, // Why: consent is per machine and the index lives with the transcripts, so // this writes the addressed host's own setting; the id never redirects it. diff --git a/src/main/runtime/runtime-ai-vault-commands.ts b/src/main/runtime/runtime-ai-vault-commands.ts index ae6d4c88656..0773e9f094d 100644 --- a/src/main/runtime/runtime-ai-vault-commands.ts +++ b/src/main/runtime/runtime-ai-vault-commands.ts @@ -35,9 +35,17 @@ import { SessionSearchQuerySchema, type SessionSearchConfigure } from '../../shared/ai-vault-search-contract' +import { + SESSION_SEARCH_METHODS, + type SessionSearchOperation +} from '../../shared/ai-vault-search-rpc-methods' export type AiVaultSessionSearchConfigureArgs = SessionSearchConfigure +// Why: `reason` is optional on the wire type, so a host that omits it must not +// surface an `undefined` message. +const SEARCH_UNAVAILABLE_MESSAGE = 'Session search is unavailable on this host.' + export class RuntimeAiVaultCommands { constructor( private readonly getPrepareResume: () => @@ -54,15 +62,18 @@ export class RuntimeAiVaultCommands { search(args: AiVaultSearchArgs, signal?: AbortSignal): Promise { const status = this.searchIndexStatus() - if (status.available === false || status.applied === false) { - throw new Error(status.reason) + // Why: an in-flight policy apply leaves `applied` false while the existing + // index is still valid; refusing there would fail every query issued during + // a settings write. `searchIndexStatus` remains the channel for that. + if (status.available === false) { + throw new Error(status.reason ?? SEARCH_UNAVAILABLE_MESSAGE) } return searchAiVaultSessions(args, { signal }).then(projectSessionSearchResult) } async sshSearch( targetId: string, - operation: 'query' | 'status' | 'configure', + operation: SessionSearchOperation, args: unknown, signal?: AbortSignal ): Promise { @@ -77,11 +88,10 @@ export class RuntimeAiVaultCommands { ? SessionSearchConfigureSchema.parse(args) : {} try { - return await provider.requestHostRpc( - `aiVault.search${{ query: 'Sessions', status: 'IndexStatus', configure: 'Configure' }[operation]}`, - params, - { signal, timeoutMs: 15_000 } - ) + return await provider.requestHostRpc(SESSION_SEARCH_METHODS[operation].relay, params, { + signal, + timeoutMs: 15_000 + }) } catch (error) { if ( operation === 'status' && @@ -129,7 +139,7 @@ export class RuntimeAiVaultCommands { } const status = this.searchIndexStatus() if (status.available === false) { - throw new Error(status.reason) + throw new Error(status.reason ?? SEARCH_UNAVAILABLE_MESSAGE) } const current = resolveAiVaultSearchSettings(store.getSettings()) const next = { diff --git a/src/main/runtime/runtime-ai-vault-search-durability.test.ts b/src/main/runtime/runtime-ai-vault-search-durability.test.ts index 340d5aeaed5..2d848d368d4 100644 --- a/src/main/runtime/runtime-ai-vault-search-durability.test.ts +++ b/src/main/runtime/runtime-ai-vault-search-durability.test.ts @@ -8,15 +8,24 @@ const apply = vi.hoisted(() => return null }) ) -vi.mock('../ai-vault-search/session-search-enablement', () => ({ - applyAiVaultSearchSettings: apply, - readAiVaultSearchIndexStatus: () => ({ +const indexStatus = vi.hoisted(() => ({ + value: { enabled: true, historyDays: null, indexSizeBytes: 0, available: true, applied: true - }) + } as Record +})) +const searchAiVaultSessions = vi.hoisted(() => vi.fn()) +vi.mock('../ai-vault-search/session-search-enablement', () => ({ + applyAiVaultSearchSettings: apply, + readAiVaultSearchIndexStatus: () => indexStatus.value +})) +vi.mock('../ai-vault/cached-session-list', () => ({ + listAiVaultSessions: vi.fn(), + readAiVaultSearchCoverage: vi.fn(), + searchAiVaultSessions })) it('does not acknowledge enabling until the durable store barrier completes', async () => { @@ -59,3 +68,43 @@ it('reports persistence failure instead of returning a successful policy acknowl ) await expect(commands.configureSearch({ enabled: true })).rejects.toThrow('disk full') }) + +it('answers from the existing index while a policy apply is still in flight', async () => { + const coverage = { + enabled: true, + sessionsIndexed: 1, + messagesIndexed: 1, + providers: [], + backfill: 'complete' as const, + filesPending: 0, + lastIndexedAt: null + } + searchAiVaultSessions.mockResolvedValue({ hits: [], route: 'and', durationMs: 1, coverage }) + const commands = new RuntimeAiVaultCommands(() => null) + indexStatus.value = { + enabled: true, + historyDays: null, + indexSizeBytes: 0, + available: true, + applied: false, + reason: 'Index policy application or persistence failed or is pending.' + } + + await expect(commands.search({ query: 'needle' })).resolves.toMatchObject({ coverage }) + expect(searchAiVaultSessions).toHaveBeenCalled() +}) + +it('refuses a query when the index is unavailable, with a message even if the host omits one', async () => { + searchAiVaultSessions.mockClear() + const commands = new RuntimeAiVaultCommands(() => null) + indexStatus.value = { + enabled: false, + historyDays: null, + indexSizeBytes: null, + available: false, + applied: false + } + + expect(() => commands.search({ query: 'needle' })).toThrow(/unavailable on this host/) + expect(searchAiVaultSessions).not.toHaveBeenCalled() +}) diff --git a/src/relay/ai-vault-handler.test.ts b/src/relay/ai-vault-handler.test.ts index eb90958d0a2..e197d64afa9 100644 --- a/src/relay/ai-vault-handler.test.ts +++ b/src/relay/ai-vault-handler.test.ts @@ -317,7 +317,8 @@ describe('AiVaultHandler', () => { hostPlatform: getRemoteHostPlatform('linux-x64'), service: { listSessions: () => Promise.reject(new Error('sidecar crashed')), - resolveSessionTitles: () => Promise.resolve({ titles: [] }) + resolveSessionTitles: () => Promise.resolve({ titles: [] }), + search: unwiredSearch } }) @@ -338,7 +339,8 @@ describe('AiVaultHandler', () => { hostPlatform: getRemoteHostPlatform('linux-x64'), service: { listSessions: () => Promise.resolve(emptyResult()), - resolveSessionTitles: () => Promise.reject(new Error('sidecar crashed')) + resolveSessionTitles: () => Promise.reject(new Error('sidecar crashed')), + search: unwiredSearch } }) @@ -365,7 +367,8 @@ describe('AiVaultHandler', () => { const error = new Error('The operation was aborted.') error.name = 'AbortError' return Promise.reject(error) - } + }, + search: unwiredSearch } }) @@ -403,10 +406,14 @@ function createTestService( signal }), resolveSessionTitles: (requests, signal) => - readAiVaultSessionTitlesFromFiles(requests, { signal }) + readAiVaultSessionTitlesFromFiles(requests, { signal }), + search: unwiredSearch } } +const unwiredSearch = (): Promise => + Promise.reject(new Error('Search is not wired in this fixture.')) + function createMockDispatcher(): { value: RelayDispatcher call: (method: string, params: Record, signal?: AbortSignal) => Promise diff --git a/src/relay/ai-vault-handler.ts b/src/relay/ai-vault-handler.ts index 518a07c0d9a..c822b2aacc2 100644 --- a/src/relay/ai-vault-handler.ts +++ b/src/relay/ai-vault-handler.ts @@ -18,6 +18,10 @@ import { parseUnameToRelayPlatform } from '../main/ssh/relay-protocol' import { relayLogLine } from './relay-diagnostic-log' import type { RelayDispatcher } from './dispatcher' import { AiVaultScanCoordinator } from '../main/ai-vault/ai-vault-scan-coordinator' +import { + SESSION_SEARCH_METHODS, + SESSION_SEARCH_OPERATIONS +} from '../shared/ai-vault-search-rpc-methods' import type { RelayAiVaultServiceApi } from './ai-vault-service-client-state' type AiVaultHandlerOptions = { @@ -55,16 +59,10 @@ export class AiVaultHandler { dispatcher.onRequest(SSH_AI_VAULT_RESOLVE_SESSION_TITLES_METHOD, (params, context) => this.resolveSessionTitles(service, params, context.signal) ) - if (service.search) { - for (const [suffix, action] of [ - ['Sessions', 'query'], - ['IndexStatus', 'status'], - ['Configure', 'configure'] - ] as const) { - dispatcher.onRequest(`aiVault.search${suffix}`, (params, context) => - service.search!(action, params, context.signal) - ) - } + for (const operation of SESSION_SEARCH_OPERATIONS) { + dispatcher.onRequest(SESSION_SEARCH_METHODS[operation].relay, (params, context) => + service.search(operation, params, context.signal) + ) } } diff --git a/src/relay/ai-vault-service-client-state.ts b/src/relay/ai-vault-service-client-state.ts index d2e0e495b89..ccc343394d0 100644 --- a/src/relay/ai-vault-service-client-state.ts +++ b/src/relay/ai-vault-service-client-state.ts @@ -4,6 +4,7 @@ import type { AiVaultSessionTitleRequest, AiVaultSessionTitlesResult } from '../shared/ai-vault-session-title' +import type { SessionSearchOperation } from '../shared/ai-vault-search-rpc-methods' import type { SshAiVaultRelayListParams } from '../shared/ssh-ai-vault-relay' import type { RemoteHostPlatform } from '../main/ssh/ssh-remote-platform' import { @@ -44,7 +45,7 @@ export function createRelayAiVaultServiceCall(args: { }): RelayAiVaultServiceCall { return { request: args.request, - lane: relayAiVaultServiceLane(args.request.operation), + lane: relayAiVaultServiceLane(args.request), signal: args.signal, forceStart: args.request.operation === 'list' && args.request.params.force === true, resolve: args.resolve, @@ -110,11 +111,7 @@ export type RelayAiVaultServiceCall = { } export type RelayAiVaultServiceApi = { - search?( - action: 'query' | 'status' | 'configure', - params: unknown, - signal?: AbortSignal - ): Promise + search(action: SessionSearchOperation, params: unknown, signal?: AbortSignal): Promise listSessions(params: SshAiVaultRelayListParams, signal?: AbortSignal): Promise resolveSessionTitles( requests: AiVaultSessionTitleRequest[], diff --git a/src/relay/ai-vault-service-client.ts b/src/relay/ai-vault-service-client.ts index 2287f9e94b3..2da582f533d 100644 --- a/src/relay/ai-vault-service-client.ts +++ b/src/relay/ai-vault-service-client.ts @@ -84,7 +84,7 @@ export class RelayAiVaultServiceClient implements RelayAiVaultServiceApi { } } - search: NonNullable = (action, params, signal) => + search: RelayAiVaultServiceApi['search'] = (action, params, signal) => this.request( { type: 'request', id: this.nextId++, operation: 'search', action, params }, signal @@ -121,7 +121,7 @@ export class RelayAiVaultServiceClient implements RelayAiVaultServiceApi { if (this.disposed || this.restartPolicy.restartScheduled) { return } - for (const lane of ['cache', 'interactive'] as const) { + for (const lane of ['cache', 'interactive', 'search'] as const) { if (this.active.has(lane)) { continue } diff --git a/src/relay/ai-vault-service-entry.ts b/src/relay/ai-vault-service-entry.ts index 0dd7284cf75..69d8fb18301 100644 --- a/src/relay/ai-vault-service-entry.ts +++ b/src/relay/ai-vault-service-entry.ts @@ -8,6 +8,7 @@ import { isRelayAiVaultServiceRequest, relayAiVaultServiceLane, type RelayAiVaultServiceChildMessage, + type RelayAiVaultServiceLane, type RelayAiVaultServiceInit, type RelayAiVaultServiceParentMessage, type RelayAiVaultServiceRequest @@ -22,8 +23,11 @@ const cancelled = new Set() const pending = new Set() const provider = createRelayAiVaultFilesystemProvider() let init: RelayAiVaultServiceInit | null = null -let cacheLane = Promise.resolve() -let interactiveLane = Promise.resolve() +const lanes: Record> = { + cache: Promise.resolve(), + interactive: Promise.resolve(), + search: Promise.resolve() +} let shuttingDown = false let searchOwner: RelaySessionSearchOwner | null = null @@ -86,7 +90,7 @@ async function shutdown(): Promise { for (const controller of controllers.values()) { controller.abort() } - await Promise.allSettled([cacheLane, interactiveLane]) + await Promise.allSettled(Object.values(lanes)) await searchOwner?.close() process.disconnect?.() } @@ -121,11 +125,8 @@ process.on('message', (raw: RelayAiVaultServiceParentMessage) => { return } pending.add(raw.id) - if (relayAiVaultServiceLane(raw.operation) === 'interactive') { - interactiveLane = interactiveLane.then(() => execute(raw)) - return - } - cacheLane = cacheLane.then(() => execute(raw)) + const lane = relayAiVaultServiceLane(raw) + lanes[lane] = lanes[lane].then(() => execute(raw)) }) process.on('disconnect', () => void shutdown()) diff --git a/src/relay/ai-vault-service-protocol.test.ts b/src/relay/ai-vault-service-protocol.test.ts new file mode 100644 index 00000000000..087fd839014 --- /dev/null +++ b/src/relay/ai-vault-service-protocol.test.ts @@ -0,0 +1,43 @@ +import { expect, it } from 'vitest' +import { + isRelayAiVaultServiceRequest, + relayAiVaultServiceLane, + type RelayAiVaultServiceRequest +} from './ai-vault-service-protocol' + +const search = (action: 'query' | 'status' | 'configure'): RelayAiVaultServiceRequest => ({ + type: 'request', + id: 1, + operation: 'search', + action, + params: {} +}) + +it('keeps a history scan and a search query off the lane that backs interactive reads', () => { + expect(relayAiVaultServiceLane({ type: 'request', id: 1, operation: 'list', params: {} })).toBe( + 'cache' + ) + expect(relayAiVaultServiceLane(search('query'))).toBe('search') + expect( + relayAiVaultServiceLane({ type: 'request', id: 1, operation: 'titles', requests: [] }) + ).toBe('interactive') + expect(relayAiVaultServiceLane(search('status'))).toBe('interactive') + expect(relayAiVaultServiceLane(search('configure'))).toBe('interactive') + expect(new Set([relayAiVaultServiceLane(search('query')), 'interactive']).size).toBe(2) +}) + +it('refuses a search request whose action is not one this build owns', () => { + expect(isRelayAiVaultServiceRequest(search('query'))).toBe(true) + expect( + isRelayAiVaultServiceRequest({ type: 'request', id: 1, operation: 'search', params: {} }) + ).toBe(false) + expect( + isRelayAiVaultServiceRequest({ + type: 'request', + id: 1, + operation: 'search', + action: 'drop', + params: {} + }) + ).toBe(false) +}) diff --git a/src/relay/ai-vault-service-protocol.ts b/src/relay/ai-vault-service-protocol.ts index 48196777a45..3050d7555f4 100644 --- a/src/relay/ai-vault-service-protocol.ts +++ b/src/relay/ai-vault-service-protocol.ts @@ -5,6 +5,10 @@ import type { } from '../shared/ai-vault-session-title' import type { SshAiVaultRelayListParams } from '../shared/ssh-ai-vault-relay' import type { RemoteHostPlatform } from '../main/ssh/ssh-remote-platform' +import { + SESSION_SEARCH_OPERATIONS, + type SessionSearchOperation +} from '../shared/ai-vault-search-rpc-methods' export const RELAY_AI_VAULT_SERVICE_PROTOCOL = 1 @@ -20,7 +24,7 @@ export type RelayAiVaultServiceRequest = type: 'request' id: number operation: 'search' - action: 'query' | 'status' | 'configure' + action: SessionSearchOperation params: unknown } | { @@ -36,14 +40,21 @@ export type RelayAiVaultServiceRequest = requests: AiVaultSessionTitleRequest[] } -export type RelayAiVaultServiceLane = 'cache' | 'interactive' -export type RelayAiVaultServiceOperation = RelayAiVaultServiceRequest['operation'] +export type RelayAiVaultServiceLane = 'cache' | 'interactive' | 'search' -/** Queries, controls and title reads must not queue behind a full history scan. */ +/** + * `list` is a full history scan and a search `query` can drive a backfill pass, + * so neither may queue ahead of the interactive lane that title reads and the + * search controls run on. Search stays correct across the split because + * `RelaySessionSearchOwner` serializes every operation it owns. + */ export function relayAiVaultServiceLane( - operation: RelayAiVaultServiceOperation + request: RelayAiVaultServiceRequest ): RelayAiVaultServiceLane { - return operation === 'list' ? 'cache' : 'interactive' + if (request.operation === 'list') { + return 'cache' + } + return request.operation === 'search' && request.action === 'query' ? 'search' : 'interactive' } export type RelayAiVaultServiceParentMessage = @@ -78,7 +89,8 @@ export function isRelayAiVaultServiceRequest(value: unknown): value is RelayAiVa Number.isSafeInteger(message.id) && (message.operation === 'list' || message.operation === 'titles' || - message.operation === 'search') + (message.operation === 'search' && + SESSION_SEARCH_OPERATIONS.includes(message.action as SessionSearchOperation))) ) } diff --git a/src/relay/session-search-owner-policy-file.ts b/src/relay/session-search-owner-policy-file.ts new file mode 100644 index 00000000000..c770fbaa6dd --- /dev/null +++ b/src/relay/session-search-owner-policy-file.ts @@ -0,0 +1,74 @@ +import { existsSync, lstatSync, readFileSync, statSync } from 'node:fs' +import { join } from 'node:path' +import { writeDurableSecureJsonFile } from '../shared/secure-file' +import { + DEFAULT_AI_VAULT_SEARCH_SETTINGS, + type AiVaultSearchSettings +} from '../shared/ai-vault-search-settings' +import { SessionSearchConfigureSchema } from '../shared/ai-vault-search-contract' + +const POLICY_FILE_VERSION = 1 +const POLICY_FILE_MAX_BYTES = 8192 + +export const SESSION_SEARCH_POLICY_RECOVERY_HINT = + 'Run `orca search --clear-index` against this host to reset the index and policy.' + +/** Refuses a path another account could have substituted for the real one. */ +export function assertOwnedSearchPath(path: string, directory: boolean): void { + const stat = lstatSync(path) + if ( + stat.isSymbolicLink() || + (directory ? !stat.isDirectory() : !stat.isFile()) || + (process.getuid && stat.uid !== process.getuid()) + ) { + throw new Error('Unsafe search owner path.') + } +} + +function policyPath(directory: string): string { + return join(directory, 'policy.json') +} + +/** Throws when the recorded policy cannot be vouched for; absent reads as consent-off. */ +export function readSessionSearchOwnerPolicy( + directory: string, + home: string +): AiVaultSearchSettings { + const file = policyPath(directory) + if (!existsSync(file)) { + return { ...DEFAULT_AI_VAULT_SEARCH_SETTINGS } + } + assertOwnedSearchPath(file, false) + if (statSync(file).size > POLICY_FILE_MAX_BYTES) { + throw new Error('Search policy exceeds its size limit.') + } + const saved = JSON.parse(readFileSync(file, 'utf8')) + if (saved.home !== home || saved.version !== POLICY_FILE_VERSION) { + throw new Error('Search source configuration changed; host policy must be reviewed.') + } + const policy = SessionSearchConfigureSchema.parse(saved.policy) + if (typeof policy.enabled !== 'boolean' || policy.historyDays === undefined) { + throw new Error('Invalid search policy.') + } + return { + enabled: policy.enabled, + historyDays: policy.historyDays, + ...(policy.paused ? { paused: true } : {}) + } +} + +export function writeSessionSearchOwnerPolicy( + directory: string, + home: string, + policy: AiVaultSearchSettings +): void { + if ( + !writeDurableSecureJsonFile(policyPath(directory), { + home, + version: POLICY_FILE_VERSION, + policy + }) + ) { + throw new Error('Could not secure the host search policy.') + } +} diff --git a/src/relay/session-search-owner-progress.test.ts b/src/relay/session-search-owner-progress.test.ts index 502a2c3c183..c6bcd965d96 100644 --- a/src/relay/session-search-owner-progress.test.ts +++ b/src/relay/session-search-owner-progress.test.ts @@ -78,11 +78,11 @@ it('finishes a slow parse before handing ownership to another relay', async () = let complete: (() => void) | undefined let parseSignal: AbortSignal | undefined vi.spyOn(candidateParser, 'parseSearchCandidates').mockImplementation( - (_store, _candidates, signal) => + (_store, _candidates, options) => new Promise((resolve) => { complete = resolve - parseSignal = signal - signal?.addEventListener('abort', () => resolve(), { once: true }) + parseSignal = options?.signal + options?.signal?.addEventListener('abort', () => resolve(), { once: true }) }) ) const state = vi.spyOn(SessionSearchStore.prototype, 'setBackfillState') diff --git a/src/relay/session-search-owner.test.ts b/src/relay/session-search-owner.test.ts index c7ad46b848a..b17d566fcc5 100644 --- a/src/relay/session-search-owner.test.ts +++ b/src/relay/session-search-owner.test.ts @@ -136,3 +136,57 @@ it('retains an initialization failure in status until successful recovery', asyn applied: true }) }) + +it('keeps its lease through an invalid query instead of dropping the lock', async () => { + const { make } = await fixture() + const first = make() + const second = make() + await first.request('configure', { enabled: true, paused: true }) + + await expect(first.request('query', { query: ' ' })).rejects.toThrow() + + await expect(second.request('configure', { enabled: false })).rejects.toThrow('in use') + expect(await first.request('status', {})).toMatchObject({ enabled: true, applied: true }) +}) + +it('still answers index-status for a policy it cannot vouch for, and a clear recovers it', async () => { + const { directory, make } = await fixture() + const first = make() + await first.request('configure', { enabled: true }) + await first.close() + await writeFile( + join(directory, 'policy.json'), + JSON.stringify({ + home: '/somewhere/else', + version: 1, + policy: { enabled: true, historyDays: null } + }) + ) + const replacement = make() + + expect(await replacement.request('status', {})).toMatchObject({ + available: true, + applied: false, + reason: expect.stringContaining('must be reviewed') + }) + await expect(replacement.request('query', { query: 'needle' })).rejects.toThrow( + 'must be reviewed' + ) + expect( + await replacement.request('configure', { enabled: false, clearIndex: true }) + ).toMatchObject({ enabled: false, applied: true }) +}) + +it('records consent durably even when applying it to the index fails', async () => { + const { directory, make } = await fixture() + const first = make() + await first.request('configure', { enabled: false }) + await first.close() + await writeFile(join(directory, 'index.sqlite'), 'not a SQLite database') + const replacement = make() + + await expect(replacement.request('configure', { enabled: true })).rejects.toThrow() + await replacement.close() + + expect(await make().request('status', {})).toMatchObject({ enabled: true }) +}) diff --git a/src/relay/session-search-owner.ts b/src/relay/session-search-owner.ts index 1d67d5e8cf3..49e1c2d1491 100644 --- a/src/relay/session-search-owner.ts +++ b/src/relay/session-search-owner.ts @@ -1,16 +1,8 @@ -import { - existsSync, - lstatSync, - mkdirSync, - readFileSync, - statSync, - closeSync, - openSync -} from 'node:fs' +import { existsSync, mkdirSync, statSync, closeSync, openSync } from 'node:fs' import { dirname, join } from 'node:path' import { homedir } from 'node:os' import Database from '../main/sqlite/sync-database' -import { hardenSecurePath, writeDurableSecureJsonFile } from '../shared/secure-file' +import { hardenSecurePath } from '../shared/secure-file' import { restrictWindowsPathSync } from '../shared/secure-path-windows-acl' import { SessionSearchService, @@ -27,6 +19,13 @@ import { SessionSearchQuerySchema, type SessionSearchConfigure } from '../shared/ai-vault-search-contract' +import { + assertOwnedSearchPath, + readSessionSearchOwnerPolicy, + writeSessionSearchOwnerPolicy, + SESSION_SEARCH_POLICY_RECOVERY_HINT +} from './session-search-owner-policy-file' +import type { SessionSearchOperation } from '../shared/ai-vault-search-rpc-methods' import { throwIfSignalAborted } from '../shared/abort-signal-reason' import { projectSessionSearchResult } from '../shared/ai-vault-search-projection' @@ -39,6 +38,7 @@ export class RelaySessionSearchOwner { private timer: NodeJS.Timeout | null = null private disposed = false private applicationError: string | undefined + private policyError: string | undefined private readonly directory: string private readonly roots: SessionSearchScanRoots @@ -58,19 +58,20 @@ export class RelaySessionSearchOwner { } } - request( - operation: 'query' | 'status' | 'configure', - raw: unknown, - signal?: AbortSignal - ): Promise { + request(operation: SessionSearchOperation, raw: unknown, signal?: AbortSignal): Promise { return this.serialize(async () => { throwIfSignalAborted(signal) if (this.disposed) { throw new Error('Search owner is closed.') } + // Validate before acquiring: a rejected `--since` must not close a warm + // service and drop the exclusion lock the next query would have to retake. + const configureArgs = + operation === 'configure' ? SessionSearchConfigureSchema.parse(raw) : undefined + const query = operation === 'query' ? SessionSearchQuerySchema.parse(raw) : undefined const capability = sessionSearchCapability() if (operation === 'status' && !this.lock) { - this.policy = this.readPolicy() + this.policy = this.readRecordedPolicy() // An existing owner's effective policy is only observable while holding the lock. if (!existsSync(join(this.directory, 'owner.sqlite'))) { return this.status(capability.available, capability.reason) @@ -87,17 +88,25 @@ export class RelaySessionSearchOwner { if (operation === 'status') { return this.status(true) } - if (operation === 'configure') { - const args = SessionSearchConfigureSchema.parse(raw) - await this.configure(args) + // A policy this host cannot vouch for may cover different sources, so the + // only operation it still permits is the clear that resets it. + if (this.policyError && !configureArgs?.clearIndex) { + throw new Error(`${this.policyError} ${SESSION_SEARCH_POLICY_RECOVERY_HINT}`) + } + if (configureArgs) { + await this.configure(configureArgs) return this.status(true) } - const query = SessionSearchQuerySchema.parse(raw) this.service ??= this.createService() - const result = await this.service.search(query, this.roots, signal) + const result = await this.service.search(query!, this.roots, signal) return projectSessionSearchResult(result) } catch (error) { - await this.release() + // Why: a caller's cancellation says nothing about the index. Only a + // failure from the service itself makes this owner's state suspect + // enough to be worth a reacquire and a policy reread. + if (!signal?.aborted) { + await this.release() + } throw error } }) @@ -113,39 +122,19 @@ export class RelaySessionSearchOwner { return result } - private readPolicy(): AiVaultSearchSettings { - const file = join(this.directory, 'policy.json') - if (!existsSync(file)) { + /** + * Records an unreadable policy instead of throwing, so `--index-status` — the + * one command a user would run to diagnose it — still answers. + */ + private readRecordedPolicy(): AiVaultSearchSettings { + try { + const policy = readSessionSearchOwnerPolicy(this.directory, this.home) + this.policyError = undefined + return policy + } catch (error) { + this.policyError = error instanceof Error ? error.message : 'Search policy is unreadable.' return { ...DEFAULT_AI_VAULT_SEARCH_SETTINGS } } - this.assertOwned(file, false) - if (statSync(file).size > 8192) { - throw new Error('Search policy exceeds its size limit.') - } - const saved = JSON.parse(readFileSync(file, 'utf8')) - if (saved.home !== this.home || saved.sources !== 1) { - throw new Error('Search source configuration changed; host policy must be reviewed.') - } - const policy = SessionSearchConfigureSchema.parse(saved.policy) - if (typeof policy.enabled !== 'boolean' || policy.historyDays === undefined) { - throw new Error('Invalid search policy.') - } - return { - enabled: policy.enabled, - historyDays: policy.historyDays, - ...(policy.paused ? { paused: true } : {}) - } - } - - private assertOwned(path: string, directory: boolean): void { - const stat = lstatSync(path) - if ( - stat.isSymbolicLink() || - (directory ? !stat.isDirectory() : !stat.isFile()) || - (process.getuid && stat.uid !== process.getuid()) - ) { - throw new Error('Unsafe search owner path.') - } } private acquire(): void { @@ -153,10 +142,10 @@ export class RelaySessionSearchOwner { return } if (existsSync(dirname(this.directory))) { - this.assertOwned(dirname(this.directory), true) + assertOwnedSearchPath(dirname(this.directory), true) } mkdirSync(this.directory, { recursive: true, mode: 0o700 }) - this.assertOwned(this.directory, true) + assertOwnedSearchPath(this.directory, true) if (process.platform === 'win32') { if (!restrictWindowsPathSync(this.directory, true)) { throw new Error('Could not secure the host search directory.') @@ -176,11 +165,11 @@ export class RelaySessionSearchOwner { throw error } } - this.assertOwned(path, false) + assertOwnedSearchPath(path, false) for (const suffix of ['', '-wal', '-shm', '-journal']) { const file = `${this.databasePath}${suffix}` if (existsSync(file)) { - this.assertOwned(file, false) + assertOwnedSearchPath(file, false) } } const lock = new Database(path, { timeout: 0 }) @@ -193,14 +182,11 @@ export class RelaySessionSearchOwner { ) } this.lock = lock - try { - this.policy = this.readPolicy() - } catch (error) { - this.lock = null - lock.close() - throw error - } - // Do not extend on traffic: a newer relay generation must get a chance to acquire. + this.policy = this.readRecordedPolicy() + // Do not extend on traffic: a newer relay generation must get a chance to + // acquire. The one exception is an active backfill pass — yieldBackfill + // waits for it rather than restarting discovery on the replacement — so the + // lease is bounded by quiet traffic, not by wall-clock time. this.timer = setTimeout(() => { void this.serialize(() => this.yieldBackfill()).catch(() => undefined) }, 5_000) @@ -214,19 +200,16 @@ export class RelaySessionSearchOwner { ...((args.paused ?? this.policy.paused) ? { paused: true } : {}) } try { + // Persist before applying: consent that is not durable must never be the + // thing an index was built under, so a crash between the two can only ever + // leave a recorded policy whose apply is retried, not an index nobody + // consented to on the next start. + writeSessionSearchOwnerPolicy(this.directory, this.home, next) + this.policy = next + this.policyError = undefined // Configuration must remain usable even when the existing index cannot be opened. this.service ??= this.createService(false) await this.service.configure(next, this.roots, { clearIndex: args.clearIndex }) - if ( - !writeDurableSecureJsonFile(join(this.directory, 'policy.json'), { - home: this.home, - sources: 1, - policy: next - }) - ) { - throw new Error('Could not secure the host search policy.') - } - this.policy = next this.applicationError = undefined } catch (error) { this.applicationError = @@ -262,12 +245,15 @@ export class RelaySessionSearchOwner { } }, 0) } + const failure = + this.applicationError ?? + (this.policyError && `${this.policyError} ${SESSION_SEARCH_POLICY_RECOVERY_HINT}`) return { ...this.policy, available, - applied: available && !this.applicationError, + applied: available && !failure, indexSizeBytes, - ...((reason ?? this.applicationError) ? { reason: reason ?? this.applicationError } : {}) + ...((reason ?? failure) ? { reason: reason ?? failure } : {}) } } diff --git a/src/renderer/src/components/right-sidebar/AiVaultPanel.tsx b/src/renderer/src/components/right-sidebar/AiVaultPanel.tsx index 7101c462cba..bb7fbfeb593 100644 --- a/src/renderer/src/components/right-sidebar/AiVaultPanel.tsx +++ b/src/renderer/src/components/right-sidebar/AiVaultPanel.tsx @@ -45,6 +45,7 @@ import { AiVaultPanelHeader } from './AiVaultPanelHeader' import { AiVaultSessionVirtualList } from './AiVaultSessionVirtualList' import { useAiVaultSessionRefresh } from './ai-vault-session-refresh' import { + aiVaultHostScopeOptionsIncludeRemote, buildAiVaultHostScopeOptions, buildRuntimeAiVaultHostScopeOptions, useAiVaultExecutionHostScope @@ -111,11 +112,7 @@ export default function AiVaultPanel(): React.JSX.Element { availableExecutionHostScopes }) const hostScopeOptions = useMemo( - () => - buildAiVaultHostScopeOptions({ - activeExecutionHostScope, - runtimeHostOptions - }), + () => buildAiVaultHostScopeOptions({ activeExecutionHostScope, runtimeHostOptions }), [activeExecutionHostScope, runtimeHostOptions] ) const activeWorktreePath = activeWorktree?.path ?? null @@ -316,6 +313,9 @@ export default function AiVaultPanel(): React.JSX.Element { // 'All' must not be narrowed; the scoped views restrict the index the same way they restrict the scan. scopePaths: scope === 'all' ? [] : scope === 'workspace' ? activeWorktreePaths : scopePaths, executionHostScope, + // Why: on desktop only remote hosts fall back to title search, so the notice about that is + // worth showing only when the panel can actually address one. + remoteHostsAvailable: aiVaultHostScopeOptionsIncludeRemote(hostScopeOptions), sessions }) diff --git a/src/renderer/src/components/right-sidebar/AiVaultPanelHeader.tsx b/src/renderer/src/components/right-sidebar/AiVaultPanelHeader.tsx index 5e8d137a20e..c88a0bb21c6 100644 --- a/src/renderer/src/components/right-sidebar/AiVaultPanelHeader.tsx +++ b/src/renderer/src/components/right-sidebar/AiVaultPanelHeader.tsx @@ -220,7 +220,7 @@ export function AiVaultPanelHeader({ ) : null} - {consent.enabled && search.localOnly ? ( + {consent.enabled && search.remoteHostsTitleOnly ? (

{translate( 'auto.components.right.sidebar.AiVaultPanel.localSearchOnly', @@ -228,14 +228,6 @@ export function AiVaultPanelHeader({ )}

) : null} - {consent.enabled && search.updating ? ( -

- {translate( - 'auto.components.right.sidebar.AiVaultPanel.searchUpdating', - 'Updating search results…' - )} -

- ) : null} {consent.enabled ? ( AiVaultSessionResumeActions /** Search results only: the matched transcript line rendered under a row. */ getSearchEvidence?: (session: AiVaultSession) => AiVaultSearchEvidence | null - getSessionResumeInChat: (session: AiVaultSession) => AiVaultResumeInChatEligibility + getSessionResumeInChat: (session: AiVaultSession) => string | null onToggleGroup: (key: string) => void onJumpToOriginalPane: (session: AiVaultSession) => void onJumpToWorktree: (worktreeId: string) => void diff --git a/src/renderer/src/components/right-sidebar/AiVaultVirtualRow.tsx b/src/renderer/src/components/right-sidebar/AiVaultVirtualRow.tsx index db33e183987..357a4aea2a1 100644 --- a/src/renderer/src/components/right-sidebar/AiVaultVirtualRow.tsx +++ b/src/renderer/src/components/right-sidebar/AiVaultVirtualRow.tsx @@ -23,7 +23,6 @@ import { canUseLocalAiVaultSessionPathActions } from './ai-vault-session-path-actions' import { canContinueAiVaultSessionInNewSession } from './ai-vault-session-continuation' -import type { AiVaultResumeInChatEligibility } from './ai-vault-session-resume-in-chat' export type AiVaultListRow = | { type: 'group'; group: AiVaultSessionGroup } @@ -76,7 +75,7 @@ export function AiVaultVirtualRow({ getSessionResumeState: (session: AiVaultSession) => AiVaultSessionResumeState getSessionResumeActions: (session: AiVaultSession) => AiVaultSessionResumeActions getSearchEvidence?: (session: AiVaultSession) => AiVaultSearchEvidence | null - getSessionResumeInChat: (session: AiVaultSession) => AiVaultResumeInChatEligibility + getSessionResumeInChat: (session: AiVaultSession) => string | null onToggleGroup: (key: string) => void onToggleSessionDetails: (sessionId: string) => void onJumpToOriginalPane: (session: AiVaultSession) => void @@ -108,7 +107,8 @@ export function AiVaultVirtualRow({ : null const resumeState = row.type === 'session' ? getSessionResumeState(row.session) : null const resumeActions = row.type === 'session' ? getSessionResumeActions(row.session) : null - const resumeInChat = row.type === 'session' ? getSessionResumeInChat(row.session) : null + const resumeInChatWorkspaceId = + row.type === 'session' ? getSessionResumeInChat(row.session) : null const continuationWorktreeId = row.type === 'session' && canContinueAiVaultSessionInNewSession(row.session, resumeState?.worktreeId) @@ -182,8 +182,8 @@ export function AiVaultVirtualRow({ : undefined } onResumeInNewChat={ - resumeInChat?.available - ? () => onResumeInNewChat(row.session, resumeInChat.workspaceId) + resumeInChatWorkspaceId + ? () => onResumeInNewChat(row.session, resumeInChatWorkspaceId) : undefined } onResumeInWorktree={() => { diff --git a/src/renderer/src/components/right-sidebar/ai-vault-host-scope.ts b/src/renderer/src/components/right-sidebar/ai-vault-host-scope.ts index 9a52609b58a..67656ae7e98 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-host-scope.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-host-scope.ts @@ -97,6 +97,15 @@ export function buildRuntimeAiVaultHostScopeOptions( }) } +/** Whether the panel can address any host other than this computer. */ +export function aiVaultHostScopeOptionsIncludeRemote( + options: readonly AiVaultHostScopeOption[] +): boolean { + return options.some( + (option) => option.id !== LOCAL_EXECUTION_HOST_ID && option.id !== ALL_EXECUTION_HOSTS_SCOPE + ) +} + export function buildAiVaultHostScopeOptions(args: { activeExecutionHostScope: ExecutionHostId | null runtimeHostOptions: readonly AiVaultHostScopeOption[] diff --git a/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.test.ts b/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.test.ts index 3b5bf574c2e..d4d5f86b45a 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.test.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.test.ts @@ -4,13 +4,19 @@ import { createElement, StrictMode, type ReactNode } from 'react' import { act, cleanup, renderHook } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { createSearchCoverageStore } from './ai-vault-search-coverage-store' -import type { AiVaultSearchCoverage } from '../../../../shared/ai-vault-search-types' +import type { + AiVaultSearchCoverage, + AiVaultSearchIndexingProgress +} from '../../../../shared/ai-vault-search-types' import { AI_VAULT_SEARCH_COVERAGE_POLL_MS, useAiVaultSearchCoveragePoll } from './ai-vault-search-coverage-poll' -function coverage(backfill: AiVaultSearchCoverage['backfill']): AiVaultSearchCoverage { +function coverage( + backfill: AiVaultSearchCoverage['backfill'], + indexing?: Partial +): AiVaultSearchCoverage { return { enabled: true, sessionsIndexed: 5, @@ -18,18 +24,42 @@ function coverage(backfill: AiVaultSearchCoverage['backfill']): AiVaultSearchCov providers: [], backfill, filesPending: 0, - lastIndexedAt: null + lastIndexedAt: null, + ...(indexing + ? { + indexing: { + phase: 'indexing', + filesProcessed: 1, + filesTotal: 10, + failures: 0, + startedAt: 0, + ...indexing + } + } + : {}) } } let searchCoverage: ReturnType +let focusListeners: (() => void)[] beforeEach(() => { vi.useFakeTimers() - searchCoverage = vi.fn().mockResolvedValue(coverage('complete')) + focusListeners = [] + searchCoverage = vi.fn().mockResolvedValue(coverage('running', { phase: 'indexing' })) Object.defineProperty(window, 'api', { configurable: true, - value: { aiVault: { searchCoverage } } + value: { + aiVault: { + searchCoverage, + onWindowFocused: (callback: () => void) => { + focusListeners.push(callback) + return () => { + focusListeners = focusListeners.filter((listener) => listener !== callback) + } + } + } + } }) }) @@ -43,38 +73,15 @@ const wrapper = ({ children }: { children: ReactNode }): React.JSX.Element => describe('useAiVaultSearchCoveragePoll', () => { it('asks for nothing while transcript search is off', () => { - const { result } = renderHook(() => useAiVaultSearchCoveragePoll(false), { wrapper }) + const { result } = renderHook(() => useAiVaultSearchCoveragePoll(false, null, 'off'), { + wrapper + }) expect(searchCoverage).not.toHaveBeenCalled() expect(result.current).toBeNull() }) - it('keeps observing a completed index so a later clear can report rebuilding', async () => { - const { result } = renderHook(() => useAiVaultSearchCoveragePoll(true), { wrapper }) - await act(async () => {}) - - expect(result.current?.sessionsIndexed).toBe(5) - const callsAfterFirstRead = searchCoverage.mock.calls.length - await act(async () => { - await vi.advanceTimersByTimeAsync(AI_VAULT_SEARCH_COVERAGE_POLL_MS * 3) - }) - expect(searchCoverage).toHaveBeenCalledTimes(callsAfterFirstRead + 3) - }) - - it('publishes consistently slow successes without overlapping polls', async () => { - searchCoverage.mockImplementation( - () => new Promise((resolve) => setTimeout(() => resolve(coverage('running')), 5_000)) - ) - const { result } = renderHook(() => useAiVaultSearchCoveragePoll(true)) - await act(async () => { - await vi.advanceTimersByTimeAsync(20_000) - }) - expect(result.current?.backfill).toBe('running') - expect(searchCoverage).toHaveBeenCalledTimes(3) - }) - it('keeps polling while the backfill is still running', async () => { - searchCoverage.mockResolvedValue(coverage('running')) - renderHook(() => useAiVaultSearchCoveragePoll(true), { wrapper }) + renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'running-owner'), { wrapper }) await act(async () => {}) const callsAfterFirstRead = searchCoverage.mock.calls.length @@ -84,9 +91,83 @@ describe('useAiVaultSearchCoveragePoll', () => { expect(searchCoverage.mock.calls.length).toBeGreaterThan(callsAfterFirstRead) }) + it('stops polling once the index reports it is up to date', async () => { + searchCoverage.mockResolvedValue(coverage('complete', { phase: 'complete' })) + const { result } = renderHook( + () => useAiVaultSearchCoveragePoll(true, null, 'complete-owner'), + { + wrapper + } + ) + await act(async () => {}) + + expect(result.current?.sessionsIndexed).toBe(5) + const callsAfterFirstRead = searchCoverage.mock.calls.length + await act(async () => { + await vi.advanceTimersByTimeAsync(AI_VAULT_SEARCH_COVERAGE_POLL_MS * 5) + }) + expect(searchCoverage).toHaveBeenCalledTimes(callsAfterFirstRead) + }) + + it('stops polling a host that reports no indexing progress at all', async () => { + searchCoverage.mockResolvedValue(coverage('complete')) + renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'legacy-owner'), { wrapper }) + await act(async () => {}) + + const callsAfterFirstRead = searchCoverage.mock.calls.length + await act(async () => { + await vi.advanceTimersByTimeAsync(AI_VAULT_SEARCH_COVERAGE_POLL_MS * 5) + }) + expect(searchCoverage).toHaveBeenCalledTimes(callsAfterFirstRead) + }) + + it('re-reads a settled index when the window is focused again', async () => { + searchCoverage.mockResolvedValue(coverage('complete', { phase: 'complete' })) + renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'focus-owner'), { wrapper }) + await act(async () => {}) + const callsAfterFirstRead = searchCoverage.mock.calls.length + + await act(async () => { + focusListeners.forEach((listener) => listener()) + }) + expect(searchCoverage).toHaveBeenCalledTimes(callsAfterFirstRead + 1) + }) + + it('resumes polling when a focus read finds the index working again', async () => { + searchCoverage.mockResolvedValue(coverage('complete', { phase: 'complete' })) + renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'refocus-owner'), { wrapper }) + await act(async () => {}) + + searchCoverage.mockResolvedValue(coverage('running', { phase: 'indexing' })) + await act(async () => { + focusListeners.forEach((listener) => listener()) + }) + const callsAfterFocus = searchCoverage.mock.calls.length + await act(async () => { + await vi.advanceTimersByTimeAsync(AI_VAULT_SEARCH_COVERAGE_POLL_MS) + }) + expect(searchCoverage.mock.calls.length).toBeGreaterThan(callsAfterFocus) + }) + + it('publishes consistently slow successes without overlapping polls', async () => { + searchCoverage.mockImplementation( + () => + new Promise((resolve) => + setTimeout(() => resolve(coverage('running', { phase: 'indexing' })), 5_000) + ) + ) + const { result } = renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'slow-owner')) + await act(async () => { + await vi.advanceTimersByTimeAsync(20_000) + }) + expect(result.current?.backfill).toBe('running') + expect(searchCoverage).toHaveBeenCalledTimes(3) + }) + it('drops the last reading when search is turned off', async () => { const { rerender, result } = renderHook( - ({ enabled }: { enabled: boolean }) => useAiVaultSearchCoveragePoll(enabled), + ({ enabled }: { enabled: boolean }) => + useAiVaultSearchCoveragePoll(enabled, null, 'toggle-owner'), { initialProps: { enabled: true }, wrapper } ) await act(async () => {}) @@ -95,6 +176,24 @@ describe('useAiVaultSearchCoveragePoll', () => { rerender({ enabled: false }) expect(result.current).toBeNull() }) + + it('publishes the coverage a search already returned instead of re-reading it', async () => { + const fromSearch = coverage('running', { phase: 'indexing', filesProcessed: 7 }) + const { rerender, result } = renderHook( + ({ latest }: { latest: AiVaultSearchCoverage | null }) => + useAiVaultSearchCoveragePoll(true, latest, 'observe-owner'), + { wrapper, initialProps: { latest: null as AiVaultSearchCoverage | null } } + ) + await act(async () => {}) + const callsAfterMount = searchCoverage.mock.calls.length + + await act(async () => { + rerender({ latest: fromSearch }) + }) + // A result arriving must publish what it carried, not spend a round trip re-asking for it. + expect(searchCoverage).toHaveBeenCalledTimes(callsAfterMount) + expect(result.current?.indexing?.filesProcessed).toBe(7) + }) }) it('drops coverage from the previous runtime immediately and polls the new owner', async () => { @@ -120,8 +219,8 @@ it('drops coverage from the previous runtime immediately and polls the new owner }) it('shares a single polling subscription between surfaces', async () => { - const first = renderHook(() => useAiVaultSearchCoveragePoll(true)) - const second = renderHook(() => useAiVaultSearchCoveragePoll(true)) + const first = renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'shared-owner')) + const second = renderHook(() => useAiVaultSearchCoveragePoll(true, null, 'shared-owner')) await act(async () => {}) expect(searchCoverage).toHaveBeenCalledTimes(1) await act(async () => { @@ -161,9 +260,59 @@ it('observes controls after mutation and ignores the outstanding older poll', as expect(searchCoverage).toHaveBeenCalledTimes(1) finishAction() await controlled - expect(store.getSnapshot().coverage?.backfill).toBe('complete') - releaseOld(coverage('running')) + expect(store.getSnapshot().coverage?.backfill).toBe('running') + releaseOld(coverage('complete')) await Promise.resolve() - expect(store.getSnapshot().coverage?.backfill).toBe('complete') + expect(store.getSnapshot().coverage?.backfill).toBe('running') + unsubscribe() +}) + +it('still reports busy when the last surface unsubscribes mid-control', async () => { + const store = createSearchCoverageStore() + const unsubscribe = store.subscribe(() => undefined) + let finishAction!: () => void + const controlled = store.control( + () => + new Promise((resolve) => { + finishAction = resolve + }) + ) + unsubscribe() + + // Why: a snapshot that forgot the running action would let a second surface start a rival one. + expect(store.getSnapshot().busy).toBe(true) + const rival = vi.fn().mockResolvedValue(undefined) + await store.control(rival) + expect(rival).not.toHaveBeenCalled() + + finishAction() + await controlled + expect(store.getSnapshot().busy).toBe(false) +}) + +it('keeps the last good reading when the only surface unmounts', async () => { + const store = createSearchCoverageStore() + const unsubscribe = store.subscribe(() => undefined) + await vi.advanceTimersByTimeAsync(0) + expect(store.getSnapshot().coverage).not.toBeNull() + unsubscribe() + // Why: resetting here made the settings panel flash "Reading index status…" on a remount. + expect(store.getSnapshot().coverage).not.toBeNull() +}) + +it('ages one indexing run from the renderer clock, not the host clock', async () => { + const store = createSearchCoverageStore() + searchCoverage.mockResolvedValue(coverage('running', { phase: 'updating', startedAt: 10 ** 12 })) + const unsubscribe = store.subscribe(() => undefined) + await vi.advanceTimersByTimeAsync(0) + const first = store.getSnapshot() + expect(first.observedAt - first.phaseSince).toBe(0) + + await vi.advanceTimersByTimeAsync(AI_VAULT_SEARCH_COVERAGE_POLL_MS * 2) + const later = store.getSnapshot() + expect(later.phaseSince).toBe(first.phaseSince) + expect(later.observedAt - later.phaseSince).toBeGreaterThanOrEqual( + AI_VAULT_SEARCH_COVERAGE_POLL_MS + ) unsubscribe() }) diff --git a/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.ts b/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.ts index d95ca3c1b76..33632ecf0c1 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-poll.ts @@ -15,6 +15,7 @@ export function useSearchIndexing(enabled: boolean, ownerKey = '') { ...snapshot, coverage: enabled ? snapshot.coverage : null, control: store.control, + observe: store.observe, refresh: store.refresh } } @@ -24,11 +25,13 @@ export function useAiVaultSearchCoveragePoll( latest: AiVaultSearchCoverage | null = null, ownerKey = '' ): AiVaultSearchCoverage | null { - const { coverage, refresh } = useSearchIndexing(enabled, ownerKey) + const { coverage, observe } = useSearchIndexing(enabled, ownerKey) + // Why: a search result carries coverage read at answer time, so publishing it is strictly + // fresher than asking the index again for what the caller is already holding. useEffect(() => { if (enabled && latest) { - void refresh() + observe(latest) } - }, [enabled, latest, refresh]) + }, [enabled, latest, observe]) return enabled ? (coverage ?? latest) : null } diff --git a/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-store.ts b/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-store.ts index 0eeceb111fa..e79b76e0fbc 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-store.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-search-coverage-store.ts @@ -3,30 +3,91 @@ import { installWindowVisibilityInterval } from '@/lib/window-visibility-interva export const AI_VAULT_SEARCH_COVERAGE_POLL_MS = 4_000 +/** Phases that advance on their own. Every other state changes only when someone acts on it. */ +const SELF_ADVANCING_PHASES: ReadonlySet = new Set([ + 'idle', + 'discovering', + 'indexing', + 'updating' +]) + type Snapshot = { coverage: AiVaultSearchCoverage | null - error: boolean busy: boolean + /** A read or a control action did not land; both read the same to the user. */ + failed: boolean + /** Renderer clock of the newest read, so callers never subtract the host's clock from ours. */ observedAt: number - unavailable: boolean + /** Renderer clock of the first read that reported the run now on screen. */ + phaseSince: number +} + +/** Identity of one indexing run, so a re-read of the same run does not restart its age. */ +function runKey(coverage: AiVaultSearchCoverage | null): string { + const indexing = coverage?.indexing + return indexing ? `${indexing.phase}:${indexing.startedAt}` : '' } /** Settings, status bar and search share a single observation per index owner. */ export function createSearchCoverageStore() { let snapshot: Snapshot = { coverage: null, - error: false, busy: false, + failed: false, observedAt: 0, - unavailable: false + phaseSince: 0 } let generation = 0 let stop: (() => void) | null = null + let unsubscribeFocus: (() => void) | null = null const listeners = new Set<() => void>() + + // Why: an index that reached a resting phase cannot change until the user acts or the app is + // refocused, so a standing interval would keep a scanner worker resident for nothing. + const shouldPoll = (): boolean => { + if (!listeners.size) { + return false + } + const phase = snapshot.coverage?.indexing?.phase + if (phase) { + return SELF_ADVANCING_PHASES.has(phase) + } + return snapshot.coverage === null && !snapshot.failed + } + + const syncPolling = (): void => { + const wanted = shouldPoll() + if (wanted === (stop !== null)) { + return + } + if (wanted) { + stop = installWindowVisibilityInterval({ + run: () => void refresh(), + intervalMs: AI_VAULT_SEARCH_COVERAGE_POLL_MS + }) + return + } + stop?.() + stop = null + } + const publish = (next: Partial): void => { snapshot = { ...snapshot, ...next } listeners.forEach((listener) => listener()) + syncPolling() } + + const record = (coverage: AiVaultSearchCoverage): void => { + const observedAt = Date.now() + const sameRun = runKey(coverage) === runKey(snapshot.coverage) + publish({ + coverage, + failed: false, + observedAt, + phaseSince: sameRun ? snapshot.phaseSince : observedAt + }) + } + let pending: Promise | null = null const refresh = (afterControl = false): Promise => { if (snapshot.busy && !afterControl) { @@ -40,11 +101,11 @@ export function createSearchCoverageStore() { try { const coverage = await window.api.aiVault.searchCoverage() if (issued === generation) { - publish({ coverage, unavailable: false, observedAt: Date.now() }) + record(coverage) } } catch { if (issued === generation) { - publish({ coverage: null, unavailable: true }) + publish({ coverage: null, failed: true }) } } })().finally(() => { @@ -59,29 +120,25 @@ export function createSearchCoverageStore() { return { getSnapshot: () => snapshot, refresh: () => refresh(), + /** Publishes a read the caller already holds; a search result carries the freshest coverage. */ + observe(coverage: AiVaultSearchCoverage): void { + if (!snapshot.busy) { + record(coverage) + } + }, subscribe(listener: () => void): () => void { listeners.add(listener) if (listeners.size === 1) { - stop = installWindowVisibilityInterval({ - run: () => void refresh(), - intervalMs: AI_VAULT_SEARCH_COVERAGE_POLL_MS - }) + unsubscribeFocus = window.api.aiVault.onWindowFocused?.(() => void refresh()) ?? null } + syncPolling() return () => { listeners.delete(listener) if (!listeners.size) { - stop?.() - stop = null - generation++ - pending = null - snapshot = { - coverage: null, - error: false, - busy: false, - observedAt: 0, - unavailable: false - } + unsubscribeFocus?.() + unsubscribeFocus = null } + syncPolling() } }, async control(action: () => Promise): Promise { @@ -90,7 +147,7 @@ export function createSearchCoverageStore() { } const controlled = ++generation pending = null - publish({ busy: true, error: false }) + publish({ busy: true, failed: false }) try { await action() if (controlled === generation) { @@ -98,7 +155,7 @@ export function createSearchCoverageStore() { } } catch { if (controlled === generation) { - publish({ error: true }) + publish({ failed: true }) } } finally { if (controlled === generation) { diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.test.tsx b/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.test.tsx new file mode 100644 index 00000000000..906da7a3c0c --- /dev/null +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.test.tsx @@ -0,0 +1,133 @@ +// @vitest-environment happy-dom +import { renderHook } from '@testing-library/react' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { AiVaultSession } from '../../../../shared/ai-vault-types' +import type { Worktree } from '../../../../shared/worktree/types' +import { useAiVaultSessionLaunchActions } from './ai-vault-session-launch-actions' +import type { AiVaultSessionResumeTargetState } from './ai-vault-session-resume' + +const mocks = vi.hoisted(() => ({ + error: vi.fn(), + success: vi.fn(), + startStructuredAgentLaunch: vi.fn(() => ({ sessionId: 's', launchResult: Promise.resolve() })), + prepare: vi.fn(async (session: AiVaultSession) => session) +})) + +vi.mock('sonner', () => ({ toast: { error: mocks.error, success: mocks.success } })) +vi.mock('@/lib/structured-agent-session-launch', () => ({ + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch +})) +vi.mock('@/lib/ai-vault-session-resume-preparation', () => ({ + prepareAiVaultSessionForResume: mocks.prepare +})) +vi.mock('@/lib/activate-ai-vault-structured-session', () => ({ + activateAiVaultStructuredSession: vi.fn() +})) +vi.mock('@/lib/launch-ai-vault-session', () => ({ launchAiVaultSessionInNewTab: vi.fn() })) +vi.mock('@/lib/ai-vault-resume-command', () => ({ + buildAiVaultResumeCopyCommandForWorktree: vi.fn(() => ''), + buildAiVaultResumeStartupForWorktree: vi.fn(() => ({})) +})) +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealFolderWorkspace: vi.fn(), + activateAndRevealWorktree: vi.fn() +})) +vi.mock('@/store', () => ({ + useAppStore: { getState: () => ({ activeWorktreeId: 'repo-1::/repo/orca' }) } +})) + +const WORKTREE_ID = 'repo-1::/repo/orca' + +function targetState(known: boolean): AiVaultSessionResumeTargetState { + const worktree = { + id: WORKTREE_ID, + repoId: 'repo-1', + displayName: 'orca', + path: '/repo/orca', + head: 'abc123', + branch: 'main', + isBare: false, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 1, + isMainWorktree: false + } as Worktree + const repo = { id: 'repo-1', path: '/repo/orca', displayName: 'orca' } as Repo + return { + folderWorkspaces: [], + projectGroups: [], + repos: known ? [repo] : [], + worktreesByRepo: known ? { 'repo-1': [worktree] } : {} + } as AiVaultSessionResumeTargetState +} + +function session(overrides: Partial = {}): AiVaultSession { + return { + id: 'claude:session-1', + agent: 'claude', + sessionId: 'session-1', + cwd: '/repo/orca', + filePath: '/home/dev/.claude/projects/-repo-orca/session-1.jsonl', + executionHostId: 'local', + messageCount: 4, + previewMessages: [], + ...overrides + } as AiVaultSession +} + +function actions(known: boolean) { + return renderHook(() => + useAiVaultSessionLaunchActions({ + activeWorktree: null, + activeWorktreeId: WORKTREE_ID, + targetState: targetState(known) + }) + ).result +} + +beforeEach(() => { + vi.clearAllMocks() +}) + +describe('handleResumeInNewChat validates its target like its siblings', () => { + it('launches into a workspace the client can place', async () => { + const result = actions(true) + result.current.handleResumeInNewChat(session(), WORKTREE_ID) + await vi.waitFor(() => expect(mocks.startStructuredAgentLaunch).toHaveBeenCalled()) + expect(mocks.startStructuredAgentLaunch).toHaveBeenCalledWith(WORKTREE_ID, 'claude', { + resumeFrom: { providerSessionId: 'session-1' } + }) + }) + + it('refuses a workspace the client does not know', async () => { + const result = actions(false) + result.current.handleResumeInNewChat(session(), WORKTREE_ID) + await Promise.resolve() + expect(mocks.startStructuredAgentLaunch).not.toHaveBeenCalled() + expect(mocks.error).toHaveBeenCalledWith('Open a workspace before resuming a session.') + }) + + it('refuses a session recorded on another host', async () => { + const result = actions(true) + result.current.handleResumeInNewChat(session({ executionHostId: 'ssh:build-box' }), WORKTREE_ID) + await Promise.resolve() + expect(mocks.startStructuredAgentLaunch).not.toHaveBeenCalled() + expect(mocks.error).toHaveBeenCalledWith( + 'This session belongs to a different host. Open a workspace on the same host to resume it.' + ) + }) + + it('ignores an agent with no structured lane', () => { + const result = actions(true) + result.current.handleResumeInNewChat(session({ agent: 'opencode' }), WORKTREE_ID) + expect(mocks.startStructuredAgentLaunch).not.toHaveBeenCalled() + expect(mocks.error).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.ts index a43694d7b8a..8bcdf1bb668 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-launch-actions.ts @@ -148,40 +148,40 @@ export function useAiVaultSessionLaunchActions({ const handleResumeInNewChat = useCallback( (session: AiVaultSession, targetWorktreeId?: string): void => { - if (!isAgentSessionHandleProvider(session.agent)) { + // Hoisted so the provider narrowing survives into the `.then` closure without a cast. + const agent = session.agent + if (!isAgentSessionHandleProvider(agent)) { return } - const worktreeId = targetWorktreeId ?? activeWorktreeId ?? activeWorktree?.id ?? null - if (!worktreeId) { - toast.error( - translate( - 'auto.components.right.sidebar.AiVaultPanel.openWorkspaceBeforeResuming', - 'Open a workspace before resuming a session.' - ) - ) + // Why: the render-time eligibility only decides whether to offer the item; the handler is + // handed a workspace id and must run the same host/workspace check as its two siblings. + const targetId = resolveAiVaultSessionLaunchTargetOrNotify({ + sessionFilePath: session.filePath, + sessionExecutionHostId: session.executionHostId, + activeWorktreeId: activeWorktreeId ?? activeWorktree?.id ?? null, + targetWorktreeId, + targetState + }) + if (!targetId) { return } // Codex rows can live under a shared legacy home; the same preparation the terminal resume // runs re-pins them, and its result is what names the conversation the host will look for. void prepareAiVaultSessionForResume(session) - .then((preparedSession) => { - const launch = startStructuredAgentLaunch( - worktreeId, - session.agent as 'claude' | 'codex', - { + .then( + (preparedSession) => + startStructuredAgentLaunch(targetId.worktreeId, agent, { resumeFrom: { providerSessionId: preparedSession.sessionId } - } - ) - return launch.launchResult - }) + }).launchResult + ) .then(() => { - if (useAppStore.getState().activeWorktreeId !== worktreeId) { - activateAiVaultResumeWorkspace(worktreeId) + if (useAppStore.getState().activeWorktreeId !== targetId.worktreeId) { + activateAiVaultResumeWorkspace(targetId.worktreeId) } }) .catch(notifyAiVaultSessionResumeInChatFailure) }, - [activeWorktree?.id, activeWorktreeId] + [activeWorktree?.id, activeWorktreeId, targetState] ) const handleContinueInNewSession = useCallback( diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat-workspace.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat-workspace.ts index 25bff1a08cc..4e6736df35e 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat-workspace.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat-workspace.ts @@ -5,19 +5,16 @@ import { } from '@/lib/agent-launch-routing' import { getLocalProjectExecutionRuntimeContext } from '@/lib/local-preflight-context' import { getExecutionHostIdForWorktree } from '@/lib/worktree-runtime-owner' -import { - readLocalRuntimeCapabilities, - readLocalRuntimeCapabilitiesOrUnknown -} from '@/runtime/local-runtime-capabilities' +import { readLocalRuntimeCapabilitiesOrUnknown } from '@/runtime/local-runtime-capabilities' import { useAppStore } from '@/store' import type { AiVaultSession } from '../../../../shared/ai-vault-types' import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' -import { STRUCTURED_AGENT_SESSION_RESUME_HISTORY_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' -import { resolveAiVaultTargetWorkspacePath } from './ai-vault-session-launch-target' import { - resolveAiVaultSessionResumeInChatEligibility, - type AiVaultResumeInChatEligibility -} from './ai-vault-session-resume-in-chat' + STRUCTURED_AGENT_SESSION_RESUME_HISTORY_RUNTIME_CAPABILITY, + type RuntimeCapability +} from '../../../../shared/protocol-version' +import { resolveAiVaultTargetWorkspacePath } from './ai-vault-session-launch-target' +import { aiVaultSessionResumeInChatWorkspaceId } from './ai-vault-session-resume-in-chat' import type { AiVaultSessionResumeState, AiVaultSessionResumeTargetState @@ -29,39 +26,33 @@ export function resolveAiVaultSessionResumeInChatForWorkspace(args: { activeWorkspaceId: string | null targetState: AiVaultSessionResumeTargetState settings: AgentLaunchRoutingInput['settings'] -}): AiVaultResumeInChatEligibility { + /** Null while the local runtime has not answered a capability probe yet. */ + hostCapabilities: readonly RuntimeCapability[] | null +}): string | null { const targetWorkspaceId = args.resumeState.usesSessionWorktree ? args.resumeState.worktreeId : (args.resumeState.worktreeId ?? args.activeWorkspaceId) - const targetWorkspacePath = targetWorkspaceId - ? resolveAiVaultTargetWorkspacePath(args.targetState, targetWorkspaceId) - : null - return resolveAiVaultSessionResumeInChatEligibility({ + if (!targetWorkspaceId || !isAgentSessionHandleProvider(args.session.agent)) { + return null + } + const state = useAppStore.getState() + const structuredRouteAvailable = + structuredAgentLaunchSupported({ + agent: args.session.agent, + settings: args.settings, + executionHostId: getExecutionHostIdForWorktree(state, targetWorkspaceId), + hostCapabilities: args.hostCapabilities, + workspaceKind: targetWorkspaceId.startsWith('folder:') ? 'folder' : 'git-worktree', + projectRuntime: getLocalProjectExecutionRuntimeContext(state, targetWorkspaceId) + }) && + (args.hostCapabilities ?? []).includes( + STRUCTURED_AGENT_SESSION_RESUME_HISTORY_RUNTIME_CAPABILITY + ) + return aiVaultSessionResumeInChatWorkspaceId({ session: args.session, targetWorkspaceId, - targetWorkspacePath, - structuredRouteAvailable: - isAgentSessionHandleProvider(args.session.agent) && - Boolean(targetWorkspaceId) && - structuredAgentLaunchSupported({ - agent: args.session.agent, - settings: args.settings, - executionHostId: getExecutionHostIdForWorktree( - useAppStore.getState(), - targetWorkspaceId as string - ), - hostCapabilities: readLocalRuntimeCapabilitiesOrUnknown(), - workspaceKind: (targetWorkspaceId as string).startsWith('folder:') - ? 'folder' - : 'git-worktree', - projectRuntime: getLocalProjectExecutionRuntimeContext( - useAppStore.getState(), - targetWorkspaceId as string - ) - }) && - readLocalRuntimeCapabilities().includes( - STRUCTURED_AGENT_SESSION_RESUME_HISTORY_RUNTIME_CAPABILITY - ) + targetWorkspacePath: resolveAiVaultTargetWorkspacePath(args.targetState, targetWorkspaceId), + structuredRouteAvailable }) } @@ -76,7 +67,10 @@ export function useAiVaultSessionResumeInChat({ activeWorkspaceId: string | null targetState: AiVaultSessionResumeTargetState settings: AgentLaunchRoutingInput['settings'] -}): (session: AiVaultSession) => AiVaultResumeInChatEligibility { +}): (session: AiVaultSession) => string | null { + // Why: read once per panel render rather than once per visible row per render, and let a + // re-probed capability list change the callback identity so rows re-evaluate. + const hostCapabilities = readLocalRuntimeCapabilitiesOrUnknown() return useCallback( (session: AiVaultSession) => resolveAiVaultSessionResumeInChatForWorkspace({ @@ -84,8 +78,9 @@ export function useAiVaultSessionResumeInChat({ resumeState: getResumeState(session), activeWorkspaceId, targetState, - settings + settings, + hostCapabilities }), - [getResumeState, activeWorkspaceId, targetState, settings] + [getResumeState, activeWorkspaceId, targetState, settings, hostCapabilities] ) } diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.test.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.test.ts index f945cdaa95a..f8765eec0fa 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.test.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.test.ts @@ -2,12 +2,10 @@ import { describe, expect, it } from 'vitest' import type { AiVaultSession } from '../../../../shared/ai-vault-types' import { aiVaultSessionCwdMatchesWorkspace, - resolveAiVaultSessionResumeInChatEligibility + aiVaultSessionResumeInChatWorkspaceId } from './ai-vault-session-resume-in-chat' -type ResumeInChatSession = Parameters< - typeof resolveAiVaultSessionResumeInChatEligibility ->[0]['session'] +type ResumeInChatSession = Parameters[0]['session'] const WORKSPACE_PATH = '/repo/orca' @@ -23,10 +21,10 @@ function session(overrides: Partial = {}): ResumeInChatSess } } -function eligibility( - overrides: Partial[0]> = {} +function workspaceId( + overrides: Partial[0]> = {} ) { - return resolveAiVaultSessionResumeInChatEligibility({ + return aiVaultSessionResumeInChatWorkspaceId({ session: session(), targetWorkspaceId: 'repo-1::/repo/orca', targetWorkspacePath: WORKSPACE_PATH, @@ -35,81 +33,66 @@ function eligibility( }) } -describe('resolveAiVaultSessionResumeInChatEligibility', () => { +describe('aiVaultSessionResumeInChatWorkspaceId', () => { it('offers the chat for a local Claude row in its own workspace', () => { - expect(eligibility()).toEqual({ available: true, workspaceId: 'repo-1::/repo/orca' }) + expect(workspaceId()).toBe('repo-1::/repo/orca') }) it.each(['hermes', 'grok', 'opencode'] as AiVaultSession['agent'][])( 'refuses %s, which has no structured lane', (agent) => { - expect(eligibility({ session: session({ agent }) })).toEqual({ - available: false, - reason: 'agent' - }) + expect(workspaceId({ session: session({ agent }) })).toBeNull() } ) - it('refuses a row already adopted into a chat before any other check', () => { + it('refuses a row already adopted into a chat', () => { // That row reopens its own chat; a second adoption is a conflict the host would refuse. expect( - eligibility({ + workspaceId({ session: { ...session(), structuredSession: { sessionId: 'claude_1', workspaceId: 'repo-1::/repo/orca' } } }) - ).toEqual({ available: false, reason: 'already-structured' }) + ).toBeNull() }) it('refuses a row recorded on a remote host', () => { - expect(eligibility({ session: session({ executionHostId: 'ssh:build-box' }) })).toEqual({ - available: false, - reason: 'remote' - }) + expect(workspaceId({ session: session({ executionHostId: 'ssh:build-box' }) })).toBeNull() }) it('refuses a row whose transcript is stored inside WSL', () => { expect( - eligibility({ + workspaceId({ session: session({ filePath: '//wsl.localhost/Ubuntu-22.04/home/dev/.claude/projects/p/session-1.jsonl' }) }) - ).toEqual({ available: false, reason: 'remote' }) + ).toBeNull() }) it('refuses a transcript that holds no conversation', () => { - expect(eligibility({ session: session({ messageCount: 0, previewMessages: [] }) })).toEqual({ - available: false, - reason: 'empty' - }) + expect(workspaceId({ session: session({ messageCount: 0, previewMessages: [] }) })).toBeNull() }) it('offers a zero-count row whose preview proves the turns exist', () => { // Some parsers only learn the turn count from metadata that may be absent. expect( - eligibility({ + workspaceId({ session: session({ messageCount: 0, previewMessages: [{ role: 'user', text: 'hello', timestamp: null }] }) }) - ).toMatchObject({ available: true }) + ).toBe('repo-1::/repo/orca') }) it('refuses when the same pair could not take the structured route for a fresh chat', () => { - expect(eligibility({ structuredRouteAvailable: false })).toEqual({ - available: false, - reason: 'workspace' - }) + expect(workspaceId({ structuredRouteAvailable: false })).toBeNull() }) it('refuses when there is no target workspace at all', () => { - expect(eligibility({ targetWorkspaceId: null })).toEqual({ - available: false, - reason: 'workspace' - }) + expect(workspaceId({ targetWorkspaceId: null })).toBeNull() }) }) @@ -117,37 +100,34 @@ describe('workspace matching, which only Claude is bound by', () => { it('refuses a Claude row whose conversation was recorded in another workspace', () => { // Claude's SDK keys transcripts by launch cwd, so resuming elsewhere silently finds nothing. expect( - eligibility({ + workspaceId({ session: session({ cwd: '/repo/other' }), targetWorkspacePath: WORKSPACE_PATH }) - ).toEqual({ available: false, reason: 'workspace' }) + ).toBeNull() }) it('refuses a Claude row that recorded no cwd', () => { - expect(eligibility({ session: session({ cwd: null }) })).toEqual({ - available: false, - reason: 'workspace' - }) + expect(workspaceId({ session: session({ cwd: null }) })).toBeNull() }) it('keeps Codex available in a different workspace, and with no recorded cwd', () => { // Codex is handed the rollout file and a cwd, so it resumes anywhere. - expect(eligibility({ session: session({ agent: 'codex', cwd: '/repo/other' }) })).toMatchObject( - { available: true } + expect(workspaceId({ session: session({ agent: 'codex', cwd: '/repo/other' }) })).toBe( + 'repo-1::/repo/orca' + ) + expect(workspaceId({ session: session({ agent: 'codex', cwd: null }) })).toBe( + 'repo-1::/repo/orca' ) - expect(eligibility({ session: session({ agent: 'codex', cwd: null }) })).toMatchObject({ - available: true - }) }) it('treats Windows spellings of one directory as the same workspace', () => { expect( - eligibility({ + workspaceId({ session: session({ cwd: 'C:\\Users\\Dev\\repo\\Orca\\' }), targetWorkspacePath: 'c:/users/dev/repo/orca' }) - ).toMatchObject({ available: true }) + ).toBe('repo-1::/repo/orca') }) }) diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.ts index 92817807f77..f387e29e907 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-resume-in-chat.ts @@ -13,17 +13,6 @@ import { type AiVaultSession } from '../../../../shared/ai-vault-types' -export type AiVaultResumeInChatBlockedReason = - | 'agent' - | 'remote' - | 'empty' - | 'already-structured' - | 'workspace' - -export type AiVaultResumeInChatEligibility = - | { available: true; workspaceId: string } - | { available: false; reason: AiVaultResumeInChatBlockedReason } - /** * Claude and Codex do not have the same freedom about *where* a conversation may be resumed. * @@ -57,7 +46,8 @@ export function aiVaultSessionCwdMatchesWorkspace( ) } -export function resolveAiVaultSessionResumeInChatEligibility(args: { +/** The workspace to resume into, or null when this row cannot be resumed into a chat at all. */ +export function aiVaultSessionResumeInChatWorkspaceId(args: { session: Pick< AiVaultSession, 'agent' | 'cwd' | 'filePath' | 'executionHostId' | 'messageCount' | 'previewMessages' @@ -68,33 +58,26 @@ export function resolveAiVaultSessionResumeInChatEligibility(args: { * re-derived: it already encodes the settings flag, host capability, platform refusals and the * WSL/repair refusal, and a second copy of those conditions would drift from it. */ structuredRouteAvailable: boolean -}): AiVaultResumeInChatEligibility { +}): string | null { const { session } = args - if (!isAgentSessionHandleProvider(session.agent)) { - return { available: false, reason: 'agent' } - } // An already-adopted row reopens its own chat instead; offering a second resume of it would ask // for a conflict the host would rightly refuse. - if (session.structuredSession) { - return { available: false, reason: 'already-structured' } - } if ( + !isAgentSessionHandleProvider(session.agent) || + session.structuredSession || session.executionHostId !== LOCAL_EXECUTION_HOST_ID || - isWslStoredAiVaultSessionFile(session.filePath) + isWslStoredAiVaultSessionFile(session.filePath) || + !isAiVaultSessionResumableContent(session) || + !args.targetWorkspaceId || + !args.structuredRouteAvailable ) { - return { available: false, reason: 'remote' } - } - if (!isAiVaultSessionResumableContent(session)) { - return { available: false, reason: 'empty' } - } - if (!args.targetWorkspaceId || !args.structuredRouteAvailable) { - return { available: false, reason: 'workspace' } + return null } if ( aiVaultSessionResumeInChatWorkspaceMatters(session.agent) && !aiVaultSessionCwdMatchesWorkspace(session.cwd, args.targetWorkspacePath) ) { - return { available: false, reason: 'workspace' } + return null } - return { available: true, workspaceId: args.targetWorkspaceId } + return args.targetWorkspaceId } diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.test.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.test.ts index 7717a077030..2c23ebf7b5a 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.test.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.test.ts @@ -172,7 +172,6 @@ describe('useAiVaultSessionSearchRequest', () => { flush: expect.any(Function), result: null, loading: false, - updating: false, error: null }) }) @@ -201,7 +200,6 @@ describe('useAiVaultSessionSearchRequest', () => { }) expect(result.current.result?.repairedTerms).toEqual(['alpha']) - expect(result.current.updating).toBe(true) expect(result.current.loading).toBe(false) }) diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.ts index 1594b5b24b3..c26ea5aeaa8 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-search-request.ts @@ -13,8 +13,6 @@ export type AiVaultSearchRequestState = { result: AiVaultSearchResult | null /** No answer for any query yet; the list should show its spinner. */ loading: boolean - /** An answer is on screen but a newer query is still in flight. */ - updating: boolean error: string | null flush: () => void } @@ -119,7 +117,6 @@ export function useAiVaultSessionSearchRequest( flush, result: current?.result ?? previous, loading: argsKey !== '' && current === null && previous === null, - updating: argsKey !== '' && !current?.full, error: current?.error ?? null } } diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.test.tsx b/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.test.tsx index 34c40d6c312..f27fc299b2a 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.test.tsx +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.test.tsx @@ -22,9 +22,8 @@ const input = { agents: ['claude' as const], scopePaths: ['/folder'], executionHostScope: 'local' as const, - sessions: [], - worktrees: [], - repos: [] + remoteHostsAvailable: false, + sessions: [] } describe('session search request boundary', () => { it('preserves operators and intersects them with the selected folder at the server', () => { @@ -44,7 +43,31 @@ describe('session search request boundary', () => { useAiVaultSessionSearchResults({ ...input, executionHostScope: 'ssh:host' }) ) expect(mocks.request.mock.lastCall?.[0]).toBeNull() - expect(result.current.localOnly).toBe(true) + expect(result.current.remoteHostsTitleOnly).toBe(true) + }) + + it('keeps the title-only notice off a desktop with nothing but this computer', () => { + const { result } = renderHook(() => useAiVaultSessionSearchResults(input)) + expect(result.current.remoteHostsTitleOnly).toBe(false) + }) + + it('warns about title-only rows once a remote host can be searched across', () => { + const { result } = renderHook(() => + useAiVaultSessionSearchResults({ + ...input, + executionHostScope: 'all', + remoteHostsAvailable: true + }) + ) + expect(result.current.remoteHostsTitleOnly).toBe(true) + }) + + it('stays quiet about remote hosts on a paired web client', () => { + mocks.web = true + const { result } = renderHook(() => + useAiVaultSessionSearchResults({ ...input, executionHostScope: 'ssh:host' }) + ) + expect(result.current.remoteHostsTitleOnly).toBe(false) }) it('keeps operator-only queries on the index path', () => { renderHook(() => useAiVaultSessionSearchResults({ ...input, query: 'path:/folder' })) diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.ts index 2c6d4b3529e..3fb6290a492 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-search-results.ts @@ -26,10 +26,9 @@ const AI_VAULT_SEARCH_PANEL_LIMIT = 50 export type AiVaultSessionSearchView = { /** True once the text half of the query is non-empty; the plain list is hidden. */ active: boolean - localOnly: boolean + /** A host the index cannot reach is in scope, so those rows match on title only. */ + remoteHostsTitleOnly: boolean loading: boolean - /** Results are on screen but a newer query is still resolving. */ - updating: boolean error: string | null coverage: AiVaultSearchCoverage | null /** Terms the index corrected before searching; empty when the query ran as typed. */ @@ -53,16 +52,24 @@ export function useAiVaultSessionSearchResults(input: { agents: readonly AiVaultAgent[] scopePaths: readonly string[] executionHostScope: ExecutionHostScope + /** Whether any host other than this computer can be chosen in the panel's host menu. */ + remoteHostsAvailable: boolean sessions: readonly AiVaultSession[] }): AiVaultSessionSearchView { const [newestFirst, setNewestFirst] = useState(false) const { agents, enabled, executionHostScope, query, scopePaths, sessions } = input - const localOnly = !isWebClientLocation() + // The desktop app searches its own index; a paired web client searches the runtime it addresses. + const isDesktopApp = !isWebClientLocation() const supportedHost = - !localOnly || + !isDesktopApp || executionHostScope === LOCAL_EXECUTION_HOST_ID || executionHostScope === ALL_EXECUTION_HOSTS_SCOPE + const remoteHostsTitleOnly = + isDesktopApp && + (executionHostScope === ALL_EXECUTION_HOSTS_SCOPE + ? input.remoteHostsAvailable + : executionHostScope !== LOCAL_EXECUTION_HOST_ID) const args = useMemo((): AiVaultSearchArgs | null => { if (!enabled || !supportedHost || agents.length === 0 || !query.trim()) { @@ -77,21 +84,18 @@ export function useAiVaultSessionSearchResults(input: { } }, [agents, enabled, newestFirst, query, scopePaths, supportedHost]) - const { error, flush, loading, result, updating } = useAiVaultSessionSearchRequest( - args, - executionHostScope - ) + const { error, flush, loading, result } = useAiVaultSessionSearchRequest(args, executionHostScope) // With an empty box no search runs, so the panel reads coverage directly to // report what is already searchable while the backfill is still going. const polledCoverage = useAiVaultSearchCoveragePoll( enabled && supportedHost, result?.coverage ?? null, - localOnly ? '' : executionHostScope + isDesktopApp ? '' : executionHostScope ) // Desktop search always reads this machine's index; a paired web client's // reads its runtime host, which is the scope it is pinned to. const executionHostId = - localOnly || executionHostScope === ALL_EXECUTION_HOSTS_SCOPE + isDesktopApp || executionHostScope === ALL_EXECUTION_HOSTS_SCOPE ? LOCAL_EXECUTION_HOST_ID : executionHostScope @@ -120,11 +124,10 @@ export function useAiVaultSessionSearchResults(input: { return useMemo( () => ({ active: args !== null, - localOnly, + remoteHostsTitleOnly, loading, - updating, error, - coverage: polledCoverage ?? result?.coverage ?? null, + coverage: polledCoverage, repairedTerms: result?.repairedTerms ?? [], disabled: isAiVaultSearchDisabled(result?.coverage), flush, @@ -142,12 +145,11 @@ export function useAiVaultSessionSearchResults(input: { groups, hitSessions, loading, - localOnly, newestFirst, polledCoverage, + remoteHostsTitleOnly, result, - sessions.length, - updating + sessions.length ] ) } diff --git a/src/renderer/src/components/settings/AgentSessionHistoryPane.tsx b/src/renderer/src/components/settings/AgentSessionHistoryPane.tsx index 613973cf358..9ee91afb6f2 100644 --- a/src/renderer/src/components/settings/AgentSessionHistoryPane.tsx +++ b/src/renderer/src/components/settings/AgentSessionHistoryPane.tsx @@ -71,14 +71,11 @@ export function AgentSessionHistoryPane({ } }, [mountedRef]) + // Why: the file only grows while the index is working, so one coverage read serves both numbers + // and the size stops being re-read the moment indexing settles. useEffect(() => { void refreshIndexSize() - if (!policy.enabled) { - return - } - const timer = setInterval(() => void refreshIndexSize(), 4000) - return () => clearInterval(timer) - }, [policy.enabled, refreshIndexSize]) + }, [indexing.observedAt, refreshIndexSize]) const apply = (next: AiVaultSearchSettings): void => { void indexing.control(() => updateSettings({ aiVaultSearch: next })) @@ -126,7 +123,7 @@ export function AgentSessionHistoryPane({ onChange={() => apply({ ...policy, enabled: !policy.enabled })} /> - {!policy.enabled && indexing.error ? ( + {!policy.enabled && indexing.failed ? (

{translate( 'sessionSearch.indexing.settingsError', diff --git a/src/renderer/src/components/settings/SessionSearchIndexingPanel.test.tsx b/src/renderer/src/components/settings/SessionSearchIndexingPanel.test.tsx index e0b1b9bc344..6e143d0ddfb 100644 --- a/src/renderer/src/components/settings/SessionSearchIndexingPanel.test.tsx +++ b/src/renderer/src/components/settings/SessionSearchIndexingPanel.test.tsx @@ -20,8 +20,7 @@ function panel(indexing?: AiVaultSearchCoverage['indexing']) { ) @@ -74,3 +73,15 @@ it('does not offer unsupported controls for legacy coverage', () => { expect(screen.queryByRole('button')).toBeNull() expect(screen.getByText('Searchable conversations: 12 · Messages: 30')).toBeTruthy() }) + +it('falls back to an indeterminate bar rather than printing 12,000 of 10,000 as 100%', () => { + panel({ + phase: 'updating', + filesProcessed: 12_000, + filesTotal: 10_000, + failures: 0, + startedAt: 0 + }) + expect(screen.getByRole('progressbar').getAttribute('aria-valuenow')).toBeNull() + expect(screen.queryByText(/Files processed/)).toBeNull() +}) diff --git a/src/renderer/src/components/settings/SessionSearchIndexingPanel.tsx b/src/renderer/src/components/settings/SessionSearchIndexingPanel.tsx index 1a6a1d8b73d..d6d6e834902 100644 --- a/src/renderer/src/components/settings/SessionSearchIndexingPanel.tsx +++ b/src/renderer/src/components/settings/SessionSearchIndexingPanel.tsx @@ -1,4 +1,7 @@ -import type { AiVaultSearchCoverage } from '../../../../shared/ai-vault-search-types' +import type { + AiVaultSearchCoverage, + AiVaultSearchIndexingProgress +} from '../../../../shared/ai-vault-search-types' import { Button } from '../ui/button' import { Progress } from '../ui/progress' import { translate } from '@/i18n/i18n' @@ -22,44 +25,54 @@ export function indexingPhaseLabel(phase: string): string { } } +/** + * Null when the total is unknown or already overtaken. Capping at 100% instead would print + * "12,000 / 10,000 · 100%"; an indeterminate bar is the honest answer to an unknown total. + */ +export function aiVaultIndexingPercentage( + progress: AiVaultSearchIndexingProgress | undefined +): number | null { + if (!progress?.filesTotal || progress.filesProcessed > progress.filesTotal) { + return null + } + return Math.floor((progress.filesProcessed / progress.filesTotal) * 100) +} + +/** A run that is moving. Paused and failed runs must not show a bar that looks like progress. */ +function indexingUnderway(phase: string | undefined): boolean { + return phase === 'discovering' || phase === 'indexing' || phase === 'updating' +} + export function SessionSearchIndexingPanel({ coverage, busy, - error, - unavailable, + failed, onControl }: { coverage: AiVaultSearchCoverage | null busy: boolean - error: boolean - unavailable: boolean + /** A read or a control action did not land. */ + failed: boolean onControl: (paused: boolean) => void }): React.JSX.Element { const progress = coverage?.indexing const phase = progress?.phase const canPause = phase !== 'paused' && phase !== 'error' && phase !== 'idle' - const percentage = - progress?.filesTotal != null && progress.filesTotal > 0 - ? Math.min(100, Math.floor((progress.filesProcessed / progress.filesTotal) * 100)) - : null - const label = unavailable - ? translate('sessionSearch.indexing.statusUnavailable', 'Index status unavailable') - : phase - ? indexingPhaseLabel(phase) - : coverage - ? indexingPhaseLabel(coverage.backfill === 'complete' ? 'complete' : 'indexing') - : translate('sessionSearch.indexing.loading', 'Reading index status…') + const percentage = aiVaultIndexingPercentage(progress) + const label = + failed && !coverage + ? translate('sessionSearch.indexing.statusUnavailable', 'Index status unavailable') + : phase + ? indexingPhaseLabel(phase) + : coverage + ? indexingPhaseLabel(coverage.backfill === 'complete' ? 'complete' : 'indexing') + : translate('sessionSearch.indexing.loading', 'Reading index status…') return (

{label} {progress ? ( - ) : null}
- {progress && phase !== 'complete' && (percentage !== null || phase === 'discovering') ? ( + {progress && phase !== 'complete' && (percentage !== null || indexingUnderway(phase)) ? ( <> {percentage !== null ? ( @@ -127,7 +140,7 @@ export function SessionSearchIndexingPanel({ )}

) : null} - {error || unavailable ? ( + {failed ? (

{translate( 'sessionSearch.indexing.unavailable', diff --git a/src/renderer/src/components/status-bar/SessionSearchStatusSegment.test.tsx b/src/renderer/src/components/status-bar/SessionSearchStatusSegment.test.tsx new file mode 100644 index 00000000000..504f9b46501 --- /dev/null +++ b/src/renderer/src/components/status-bar/SessionSearchStatusSegment.test.tsx @@ -0,0 +1,103 @@ +// @vitest-environment happy-dom +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { TooltipProvider } from '../ui/tooltip' +import { SessionSearchStatusSegment } from './SessionSearchStatusSegment' +import type { AiVaultSearchIndexingProgress } from '../../../../shared/ai-vault-search-types' + +const mocks = vi.hoisted(() => ({ + indexing: { + coverage: null as unknown, + busy: false, + failed: false, + observedAt: 0, + phaseSince: 0 + }, + openSettingsPage: vi.fn(), + setSettingsSearchQuery: vi.fn() +})) + +vi.mock('../right-sidebar/ai-vault-search-coverage-poll', () => ({ + AI_VAULT_SEARCH_COVERAGE_POLL_MS: 4_000, + useSearchIndexing: () => mocks.indexing +})) +vi.mock('@/store', () => ({ + useAppStore: Object.assign(() => ({ aiVaultSearch: { enabled: true } }), { + getState: () => ({ + openSettingsPage: mocks.openSettingsPage, + setSettingsSearchQuery: mocks.setSettingsSearchQuery + }) + }) +})) + +afterEach(cleanup) +beforeEach(() => { + vi.clearAllMocks() +}) + +function segment( + progress: Partial | null, + observed: { observedAt: number; phaseSince: number } = { observedAt: 10_000, phaseSince: 0 } +) { + mocks.indexing = { + ...mocks.indexing, + ...observed, + coverage: progress + ? { + enabled: true, + sessionsIndexed: 1, + messagesIndexed: 1, + providers: [], + backfill: 'running', + filesPending: 0, + lastIndexedAt: null, + indexing: { + phase: 'indexing', + filesProcessed: 3, + filesTotal: 10, + failures: 0, + startedAt: 0, + ...progress + } + } + : null + } + render( + + + + ) +} + +it('reports the phase and how far the index has got', () => { + segment({ phase: 'indexing', filesProcessed: 3, filesTotal: 10 }) + expect(screen.getByRole('button', { name: 'Indexing conversations · 30%' })).toBeTruthy() +}) + +it('stays out of the status bar once the index is up to date', () => { + segment({ phase: 'complete' }) + expect(screen.queryByRole('button')).toBeNull() +}) + +it('waits out an incremental pass that may finish within one poll', () => { + segment({ phase: 'updating' }, { observedAt: 1_000, phaseSince: 0 }) + expect(screen.queryByRole('button')).toBeNull() +}) + +it('shows an incremental pass that has outlived a poll interval', () => { + segment({ phase: 'updating' }, { observedAt: 9_000, phaseSince: 0 }) + expect(screen.queryByRole('button')).not.toBeNull() +}) + +it('offers no controls of its own; the settings pane owns them', () => { + segment({ phase: 'paused' }) + expect(screen.queryByRole('button', { name: 'Resume' })).toBeNull() + expect(screen.getAllByRole('button')).toHaveLength(1) +}) + +it('opens the settings pane that owns the controls', () => { + segment({ phase: 'paused' }) + fireEvent.click(screen.getByRole('button')) + expect(mocks.openSettingsPage).toHaveBeenCalled() + expect(mocks.setSettingsSearchQuery).toHaveBeenCalledWith('Agent Session History') +}) diff --git a/src/renderer/src/components/status-bar/SessionSearchStatusSegment.tsx b/src/renderer/src/components/status-bar/SessionSearchStatusSegment.tsx index 9700d1e8d0e..9d3b7c8ff62 100644 --- a/src/renderer/src/components/status-bar/SessionSearchStatusSegment.tsx +++ b/src/renderer/src/components/status-bar/SessionSearchStatusSegment.tsx @@ -1,16 +1,21 @@ import { AlertCircle, Loader2, Pause } from 'lucide-react' import { useAppStore } from '@/store' -import { useSearchIndexing } from '../right-sidebar/ai-vault-search-coverage-poll' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { - SessionSearchIndexingPanel, + useSearchIndexing, + AI_VAULT_SEARCH_COVERAGE_POLL_MS +} from '../right-sidebar/ai-vault-search-coverage-poll' +import { + aiVaultIndexingPercentage, indexingPhaseLabel } from '../settings/SessionSearchIndexingPanel' import { resolveAiVaultSearchSettings } from '../../../../shared/ai-vault-search-settings' -import { Popover, PopoverContent, PopoverTrigger } from '../ui/popover' -import { Button } from '../ui/button' import { translate } from '@/i18n/i18n' -import { STATUS_BAR_CONTEXT_MENU_EXEMPT_PROPS } from './status-bar-context-menu-policy' +/** + * Read-only: the settings pane owns pausing, resuming and retrying. This segment says only that + * search results are still partial, and takes the user to the controls. + */ export function SessionSearchStatusSegment({ iconOnly }: { @@ -20,60 +25,55 @@ export function SessionSearchStatusSegment({ const policy = resolveAiVaultSearchSettings(settings) const indexing = useSearchIndexing(policy.enabled) const progress = indexing.coverage?.indexing - if ( - !progress || - progress.phase === 'complete' || - (progress.phase === 'updating' && indexing.observedAt - progress.startedAt < 4000) - ) { + // Why: an incremental pass that finishes inside one poll interval would only flicker. Both + // timestamps are renderer clocks, so this stays meaningful when the index runs on another host. + const settling = + progress?.phase === 'updating' && + indexing.observedAt - indexing.phaseSince < AI_VAULT_SEARCH_COVERAGE_POLL_MS + if (!progress || progress.phase === 'complete' || settling) { return null } - const label = indexingPhaseLabel(progress.phase) + const percentage = aiVaultIndexingPercentage(progress) + const phase = indexingPhaseLabel(progress.phase) + const label = + percentage === null + ? phase + : translate('sessionSearch.indexing.segmentProgress', '{{phase}} · {{percent}}%', { + phase, + percent: percentage + }) const Icon = progress.phase === 'paused' ? Pause : progress.phase === 'error' ? AlertCircle : Loader2 return ( - - + + - - - { - void indexing.control(() => - useAppStore.getState().updateSettingsOrThrow({ aiVaultSearch: { ...policy, paused } }) - ) - }} - /> - - - + + + {translate( + 'sessionSearch.indexing.segmentTooltip', + '{{status}}. Click to open search settings.', + { status: label } + )} + + ) } diff --git a/src/renderer/src/components/ui/progress.tsx b/src/renderer/src/components/ui/progress.tsx index bb5ea77f90f..4f35fbe764b 100644 --- a/src/renderer/src/components/ui/progress.tsx +++ b/src/renderer/src/components/ui/progress.tsx @@ -19,7 +19,8 @@ function Progress({ data-slot="progress-indicator" className={cn( 'h-full w-full flex-1 bg-primary transition-all duration-300 ease-out', - value === null && 'w-1/3 animate-pulse' + // Why: an indeterminate bar must still read as indeterminate without motion. + value === null && 'w-1/3 animate-pulse motion-reduce:animate-none' )} style={value === null ? undefined : { transform: `translateX(-${100 - (value || 0)}%)` }} /> diff --git a/src/renderer/src/i18n/en-runtime-required.json b/src/renderer/src/i18n/en-runtime-required.json index 9dfbdd11ca8..0e25329f8ce 100644 --- a/src/renderer/src/i18n/en-runtime-required.json +++ b/src/renderer/src/i18n/en-runtime-required.json @@ -683,9 +683,9 @@ "localSessionSshWorkspaceUnsupported": "This session's history is stored on this machine, so it can't resume in an SSH workspace. Open a local workspace instead.", "localWorkspacesOnly": "Resume from history is only available in local workspaces.", "remoteBrowseLocalHistory": "Remote workspaces can browse local history. Resume actions run from local workspaces.", - "valueCopyFailed": "Unable to copy {{value0}}", "searchMatches_one": "{{count}} match", - "searchMatches_other": "{{count}} matches" + "searchMatches_other": "{{count}} matches", + "valueCopyFailed": "Unable to copy {{value0}}" }, "AiVaultPanelControls": { "allSessions": "All sessions", diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 719751c3167..63761760095 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -13064,7 +13064,6 @@ "searchMatches_other": "{{count}} matches", "searchMatches": "{{count}} matches", "localSearchOnly": "Conversation search covers this computer only. Remote hosts use title search.", - "searchUpdating": "Updating search results…", "resumeInChatConflict": "Another chat is already holding this conversation.", "resumeInChatTranscriptMissing": "This conversation's history could not be loaded, so it cannot be resumed in chat.", "resumeInChatFailed": "Could not resume this session in a new chat." @@ -17820,7 +17819,8 @@ "counts": "Searchable conversations: {{sessions}} · Messages: {{messages}}", "failures": "{{count}} indexing issues. Retry to check the remaining files.", "unavailable": "Could not update index status. Please try again.", - "openSettings": "Open settings", + "segmentProgress": "{{phase}} · {{percent}}%", + "segmentTooltip": "{{status}}. Click to open search settings.", "settingsError": "Could not save search settings. Please try again.", "idle": "Waiting to index", "start": "Index now" diff --git a/src/shared/ai-vault-search-contract.ts b/src/shared/ai-vault-search-contract.ts index 687aba64aa3..41cf35bca80 100644 --- a/src/shared/ai-vault-search-contract.ts +++ b/src/shared/ai-vault-search-contract.ts @@ -35,31 +35,41 @@ export const SessionSearchStatusSchema = z.object({ }) const boundedText = z.string().max(32768) + +/** + * Why the received-payload enums fall back instead of failing: a client and the + * host it queries update independently, so a host that learns one new route, + * indexing phase, or message role must not cost an older client the whole + * response. Each fallback is the value that already means "nothing specific". + */ +export const SessionSearchHitSchema = z.object({ + agent: z.enum(AI_VAULT_AGENTS), + sessionId: boundedText, + filePath: boundedText, + codexHome: boundedText.nullable(), + title: boundedText, + cwd: boundedText.nullable(), + branch: boundedText.nullable(), + updatedAt: boundedText.nullable(), + messageCount: z.number().nonnegative(), + resumeCommand: boundedText, + score: z.number(), + duplicateCount: z.number().optional(), + evidence: z.object({ + role: z.enum(['user', 'assistant', 'tool', 'system', 'unknown']).catch('unknown'), + timestamp: boundedText.nullable(), + snippet: boundedText + }) +}) + export const SessionSearchResultSchema = z.object({ + // Why per-hit and not per-response: an agent the client cannot name has no + // resume command it could run, so that hit is the only thing it should lose. hits: z - .array( - z.object({ - agent: z.enum(AI_VAULT_AGENTS), - sessionId: boundedText, - filePath: boundedText, - codexHome: boundedText.nullable(), - title: boundedText, - cwd: boundedText.nullable(), - branch: boundedText.nullable(), - updatedAt: boundedText.nullable(), - messageCount: z.number().nonnegative(), - resumeCommand: boundedText, - score: z.number(), - duplicateCount: z.number().optional(), - evidence: z.object({ - role: z.enum(['user', 'assistant', 'tool', 'system', 'unknown']), - timestamp: boundedText.nullable(), - snippet: boundedText - }) - }) - ) - .max(AI_VAULT_SEARCH_LIMIT_MAX), - route: z.enum(['phrase', 'and', 'or', 'typo+phrase', 'typo+and', 'typo+or']), + .array(SessionSearchHitSchema.nullable().catch(null)) + .max(AI_VAULT_SEARCH_LIMIT_MAX) + .transform((hits) => hits.filter((hit) => hit !== null)), + route: z.enum(['phrase', 'and', 'or', 'typo+phrase', 'typo+and', 'typo+or']).catch('or'), repairedTerms: z.array(boundedText).max(512).optional(), durationMs: z.number().nonnegative(), coverage: z.object({ @@ -78,20 +88,17 @@ export const SessionSearchResultSchema = z.object({ }) ) .max(AI_VAULT_AGENTS.length), - backfill: z.enum(['idle', 'running', 'complete']), + // An unknown backfill state is not evidence the index is finished. + backfill: z.enum(['idle', 'running', 'complete']).catch('running'), filesPending: z.number().nonnegative(), lastIndexedAt: boundedText.nullable(), indexing: z .object({ - phase: z.enum([ - 'idle', - 'discovering', - 'indexing', - 'updating', - 'paused', - 'complete', - 'error' - ]), + // Unknown phases report as work in flight; `idle` would claim the + // opposite of what a newer host is telling us. + phase: z + .enum(['idle', 'discovering', 'indexing', 'updating', 'paused', 'complete', 'error']) + .catch('indexing'), filesProcessed: z.number().nonnegative(), filesTotal: z.number().nonnegative().nullable(), failures: z.number().nonnegative(), diff --git a/src/shared/ai-vault-search-coverage.ts b/src/shared/ai-vault-search-coverage.ts index ff55199d408..e944604e964 100644 --- a/src/shared/ai-vault-search-coverage.ts +++ b/src/shared/ai-vault-search-coverage.ts @@ -1,4 +1,8 @@ -import type { AiVaultSearchCoverage, AiVaultSearchProviderCoverage } from './ai-vault-search-types' +import type { + AiVaultSearchCoverage, + AiVaultSearchProviderCoverage, + AiVaultSearchResult +} from './ai-vault-search-types' /** What a host reports while transcript search is switched off. */ export const DISABLED_AI_VAULT_SEARCH_COVERAGE: AiVaultSearchCoverage = { @@ -11,6 +15,14 @@ export const DISABLED_AI_VAULT_SEARCH_COVERAGE: AiVaultSearchCoverage = { lastIndexedAt: null } +/** What a query answers with when there is no index to read: off, closing, or closed. */ +export const NO_AI_VAULT_SEARCH_INDEX_RESULT: AiVaultSearchResult = { + hits: [], + route: 'and', + durationMs: 0, + coverage: DISABLED_AI_VAULT_SEARCH_COVERAGE +} + /** Old hosts omit the flag; only an explicit `false` means the user opted out. */ export function isAiVaultSearchDisabled( coverage: Pick | null | undefined diff --git a/src/shared/ai-vault-search-projection.test.ts b/src/shared/ai-vault-search-projection.test.ts index a6af27379f8..a908950e954 100644 --- a/src/shared/ai-vault-search-projection.test.ts +++ b/src/shared/ai-vault-search-projection.test.ts @@ -1,5 +1,7 @@ -import { expect, it } from 'vitest' +import type { z } from 'zod' +import { expect, expectTypeOf, it } from 'vitest' import { projectSessionSearchResult } from './ai-vault-search-projection' +import { SessionSearchResultSchema } from './ai-vault-search-contract' import type { AiVaultSearchResult, AiVaultSearchHit } from './ai-vault-search-types' const hit: AiVaultSearchHit = { @@ -48,3 +50,53 @@ it('omits oversized identities and caps serialized response size with explicit a expect(Buffer.byteLength(JSON.stringify(projected))).toBeLessThanOrEqual(512 * 1024) expect(projected.omittedHits! + projected.hits.length).toBe(101) }) + +it('never truncates a snippet into an unclosable mark', () => { + const cutInsideAMark = { + ...hit, + evidence: { ...hit.evidence, snippet: `${'a'.repeat(4094)}[[needle]]` } + } + const snippet = projectSessionSearchResult(result([cutInsideAMark])).hits[0]!.evidence.snippet + expect(snippet).not.toContain('[[') + expect(snippet).toBe('a'.repeat(4094)) +}) + +// A client and the host it queries update independently, so one unknown enum +// value must cost at most the hit that carries it. +it('keeps a response a newer host filled with values this build does not know', () => { + const fromNewerHost: unknown = { + ...result([]), + hits: [ + { ...hit, evidence: { ...hit.evidence, role: 'planner', snippet: '' } }, + { ...hit, agent: 'agent-from-the-future', evidence: { ...hit.evidence, snippet: '' } } + ], + route: 'vector', + coverage: { + sessionsIndexed: 0, + messagesIndexed: 0, + providers: [], + backfill: 'draining', + filesPending: 0, + lastIndexedAt: null, + indexing: { + phase: 'compacting', + filesProcessed: 0, + filesTotal: null, + failures: 0, + startedAt: 0 + } + } + } + const parsed = SessionSearchResultSchema.parse(fromNewerHost) + expect(parsed.hits.map((entry) => entry.evidence.role)).toEqual(['unknown']) + expect(parsed.route).toBe('or') + expect(parsed.coverage.backfill).toBe('running') + expect(parsed.coverage.indexing?.phase).toBe('indexing') +}) + +// Why both directions: a field only on the type is stripped before transport, +// and a field only in the schema rejects a successful local search on arrival. +it('keeps the wire schema and the result type in step', () => { + expectTypeOf().toExtend>() + expectTypeOf>().toExtend() +}) diff --git a/src/shared/ai-vault-search-projection.ts b/src/shared/ai-vault-search-projection.ts index f3097ae53cb..4cd8157cee4 100644 --- a/src/shared/ai-vault-search-projection.ts +++ b/src/shared/ai-vault-search-projection.ts @@ -1,18 +1,30 @@ -import { SessionSearchResultSchema } from './ai-vault-search-contract' +import { SessionSearchHitSchema, SessionSearchResultSchema } from './ai-vault-search-contract' import type { AiVaultSearchResult } from './ai-vault-search-types' +import { + AI_VAULT_SEARCH_LIMIT_MAX, + AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE, + AI_VAULT_SEARCH_SNIPPET_MARK_OPEN +} from './ai-vault-search-types' + +const MAX_SNIPPET_BYTES = 4096 +const MAX_RESPONSE_BYTES = 512 * 1024 +// Metadata alone (coverage, counters) must leave room for at least some hits. +const MAX_METADATA_BYTES = 64 * 1024 +// The `{"hits":[ ... ]}` framing the per-hit sizes below do not include. +const RESPONSE_ENVELOPE_BYTES = 256 const encoder = new TextEncoder() const decoder = new TextDecoder('utf-8', { fatal: true }) function snippet(text: string): string { const bytes = encoder.encode(text) - if (bytes.length <= 4096) { + if (bytes.length <= MAX_SNIPPET_BYTES) { return text } - let end = 4096 + let end = MAX_SNIPPET_BYTES while (end > 0) { try { - return decoder.decode(bytes.subarray(0, end)) + return balanceMarks(decoder.decode(bytes.subarray(0, end))) } catch { end-- } @@ -20,14 +32,22 @@ function snippet(text: string): string { return '' } +/** A cut between `[[` and `]]` hands the renderer a mark it can never close. */ +function balanceMarks(text: string): string { + const opened = text.lastIndexOf(AI_VAULT_SEARCH_SNIPPET_MARK_OPEN) + return opened !== -1 && !text.includes(AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE, opened) + ? text.slice(0, opened) + : text +} + /** Bound before every host transport; never shorten session identities or resume paths. */ export function projectSessionSearchResult(result: AiVaultSearchResult): AiVaultSearchResult { const metadata = SessionSearchResultSchema.parse({ ...result, hits: [] }) const hits: AiVaultSearchResult['hits'] = [] let truncatedSnippets = result.truncatedSnippets ?? 0 let omittedHits = result.omittedHits ?? 0 - let bytes = encoder.encode(JSON.stringify(metadata)).length + 256 - if (bytes > 64 * 1024) { + let bytes = encoder.encode(JSON.stringify(metadata)).length + RESPONSE_ENVELOPE_BYTES + if (bytes > MAX_METADATA_BYTES) { throw new Error('Search metadata exceeds the response limit.') } for (const hit of result.hits) { @@ -35,9 +55,9 @@ export function projectSessionSearchResult(result: AiVaultSearchResult): AiVault const projected = { ...hit, evidence: { ...hit.evidence, snippet: text } } const size = encoder.encode(JSON.stringify(projected)).length + 1 if ( - hits.length >= 100 || - bytes + size > 512 * 1024 || - !SessionSearchResultSchema.shape.hits.element.safeParse(projected).success + hits.length >= AI_VAULT_SEARCH_LIMIT_MAX || + bytes + size > MAX_RESPONSE_BYTES || + !SessionSearchHitSchema.safeParse(projected).success ) { omittedHits++ continue diff --git a/src/shared/ai-vault-search-query-operators.test.ts b/src/shared/ai-vault-search-query-operators.test.ts index 51f33f2d84a..b476db908d4 100644 --- a/src/shared/ai-vault-search-query-operators.test.ts +++ b/src/shared/ai-vault-search-query-operators.test.ts @@ -5,6 +5,7 @@ describe('splitAiVaultSearchQuery', () => { it('sends plain text through untouched', () => { expect(splitAiVaultSearchQuery('strict mode violation')).toEqual({ text: 'strict mode violation', + terms: ['strict', 'mode', 'violation'], repoTerms: [], pathTerms: [] }) @@ -23,6 +24,12 @@ describe('splitAiVaultSearchQuery', () => { expect(split.pathTerms).toEqual(['/Users/ada/My Project']) }) + it('keeps a quoted free-text span whole and preserves its quotes for FTS', () => { + const split = splitAiVaultSearchQuery('"resume picker" repo:orca') + expect(split.text).toBe('"resume picker"') + expect(split.terms).toEqual(['resume picker']) + }) + it('reports empty text when only operators were typed', () => { expect(splitAiVaultSearchQuery('repo:orca').text).toBe('') }) @@ -32,11 +39,28 @@ describe('splitAiVaultSearchQuery', () => { expect(split.repoTerms).toEqual([]) expect(split.text).toBe('orca') }) + + // A contraction's apostrophe used to open a quoted span that swallowed the operator. + it('reads an operator between two contractions', () => { + const split = splitAiVaultSearchQuery("it's a repo:orca thing's") + expect(split.repoTerms).toEqual(['orca']) + expect(split.text).toBe("it's a thing's") + }) + + it('preserves operator value case so the path key decides folding', () => { + expect(splitAiVaultSearchQuery('path:C:\\Work\\App needle').pathTerms).toEqual([ + 'C:\\Work\\App' + ]) + expect(splitAiVaultSearchQuery('repo:MyRepo').repoTerms).toEqual(['MyRepo']) + }) }) -it.each(['myrepo:orca', 'https://host/path:word', '"path:/literal phrase"'])( +it.each(['myrepo:orca', 'https://host/path:word', '"path:/literal phrase"', "don't"])( 'preserves literal %s', (query) => { - expect(splitAiVaultSearchQuery(query)).toEqual({ text: query, repoTerms: [], pathTerms: [] }) + const split = splitAiVaultSearchQuery(query) + expect(split.text).toBe(query) + expect(split.repoTerms).toEqual([]) + expect(split.pathTerms).toEqual([]) } ) diff --git a/src/shared/ai-vault-search-query-operators.ts b/src/shared/ai-vault-search-query-operators.ts index b91ae640892..ff30b5d0fb1 100644 --- a/src/shared/ai-vault-search-query-operators.ts +++ b/src/shared/ai-vault-search-query-operators.ts @@ -1,45 +1,80 @@ -const QUERY_OPERATOR_PATTERN = /"[^"]*"|'[^']*'|(^|\s)(repo|path):(?:"([^"]*)"|'([^']*)'|(\S*))/gi +/** Anchored at a token start only, so `myrepo:x` and `https://h/path:x` stay literal. */ +const OPERATOR = /(repo|path):/iy export type AiVaultSearchQuerySplit = { + /** Query minus the operator tokens, quoting intact; what FTS sees. */ text: string + /** The same free text as tokens with quotes stripped; what a substring matcher wants. */ + terms: readonly string[] + /** Operator values exactly as typed: the panel folds case, the index does not. */ repoTerms: readonly string[] pathTerms: readonly string[] } -/** Preserve path spelling and remove only complete operator tokens. */ +/** + * The one reading of `repo:` / `path:` in the product: the sessions panel and the + * search index must agree on what is an operator and what is ordinary text. + */ export function splitAiVaultSearchQuery(query: string): AiVaultSearchQuerySplit { + const spans: string[] = [] + const terms: string[] = [] const repoTerms: string[] = [] const pathTerms: string[] = [] - const text = query - .replaceAll( - QUERY_OPERATOR_PATTERN, - ( - _match, - space: string, - operator: string | undefined, - doubleQuoted: string | undefined, - singleQuoted: string | undefined, - bare: string | undefined - ) => { - if (!operator) { - return _match - } - const value = doubleQuoted ?? singleQuoted ?? bare ?? '' - if (value) { - if (operator.toLowerCase() === 'repo') { - repoTerms.push(value.toLowerCase()) - } else { - pathTerms.push(value) - } - } - return space + let index = 0 + while (index < query.length) { + if (isBoundary(query[index])) { + index += 1 + continue + } + OPERATOR.lastIndex = index + const operator = OPERATOR.exec(query) + if (operator) { + const at = index + operator[0].length + const quoted = readQuoted(query, at) + const value = quoted?.value ?? readBare(query, at) + index = quoted ? quoted.end : at + value.length + if (value) { + ;(operator[1]!.toLowerCase() === 'repo' ? repoTerms : pathTerms).push(value) } - ) - .replace(/\s+/g, ' ') - .trim() - return { text, repoTerms, pathTerms } + continue + } + const quoted = readQuoted(query, index) + const value = quoted?.value ?? readBare(query, index) + const end = quoted ? quoted.end : index + value.length + spans.push(query.slice(index, end)) + terms.push(value) + index = end + } + return { text: spans.join(' '), terms, repoTerms, pathTerms } } export function hasAiVaultSearchQueryOperators(split: AiVaultSearchQuerySplit): boolean { return split.repoTerms.length > 0 || split.pathTerms.length > 0 } + +function isBoundary(char: string | undefined): boolean { + return char === undefined || /\s/.test(char) +} + +/** + * Why the closing quote must end a word: otherwise the apostrophes in + * `it's a repo:orca thing's` open a span that swallows the operator between them. + */ +function readQuoted(query: string, at: number): { value: string; end: number } | null { + const quote = query[at] + if (quote !== '"' && quote !== "'") { + return null + } + const close = query.indexOf(quote, at + 1) + return close === -1 || !isBoundary(query[close + 1]) + ? null + : { value: query.slice(at + 1, close), end: close + 1 } +} + +function readBare(query: string, at: number): string { + let end = at + while (end < query.length && !isBoundary(query[end])) { + end += 1 + } + return query.slice(at, end) +} diff --git a/src/shared/ai-vault-search-rpc-methods.ts b/src/shared/ai-vault-search-rpc-methods.ts new file mode 100644 index 00000000000..2afbd436c03 --- /dev/null +++ b/src/shared/ai-vault-search-rpc-methods.ts @@ -0,0 +1,29 @@ +export const SESSION_SEARCH_OPERATIONS = ['query', 'status', 'configure'] as const +export type SessionSearchOperation = (typeof SESSION_SEARCH_OPERATIONS)[number] + +/** + * One record per operation across the three method namespaces it travels. + * They are not the same names — `configure` is `aiVault.searchConfigure` on the + * relay and `aiVault.configureSessionSearch` on a runtime — so registering and + * calling from here is what keeps a registered name and a called name in step. + */ +export const SESSION_SEARCH_METHODS = { + query: { + relay: 'aiVault.searchSessions', + runtime: 'aiVault.searchSessions', + runtimeSsh: 'aiVault.sshSearchSessions' + }, + status: { + relay: 'aiVault.searchIndexStatus', + runtime: 'aiVault.searchIndexStatus', + runtimeSsh: 'aiVault.sshSearchIndexStatus' + }, + configure: { + relay: 'aiVault.searchConfigure', + runtime: 'aiVault.configureSessionSearch', + runtimeSsh: 'aiVault.sshSearchConfigure' + } +} as const satisfies Record< + SessionSearchOperation, + { relay: string; runtime: string; runtimeSsh: string } +> diff --git a/src/shared/ai-vault-search-settings.ts b/src/shared/ai-vault-search-settings.ts index f2bdb9a7046..5c55d259f03 100644 --- a/src/shared/ai-vault-search-settings.ts +++ b/src/shared/ai-vault-search-settings.ts @@ -61,14 +61,6 @@ export function aiVaultSearchHistoryCutoffMs( return historyDays === null ? null : now - historyDays * 86_400_000 } -/** null (all history) is the widest bound; otherwise more days means wider. */ -export function widensAiVaultSearchHistory(previous: number | null, next: number | null): boolean { - if (previous === null) { - return false - } - return next === null || next > previous -} - /** A finite bound tighter than before; rows outside it are purged. */ export function narrowsAiVaultSearchHistory(previous: number | null, next: number | null): boolean { if (next === null) { diff --git a/src/shared/ai-vault-session-filters.ts b/src/shared/ai-vault-session-filters.ts index 46fafbc8efd..851d0edffbd 100644 --- a/src/shared/ai-vault-session-filters.ts +++ b/src/shared/ai-vault-session-filters.ts @@ -8,6 +8,7 @@ import { normalizeRuntimePathSeparators } from './cross-platform-path' import { isClipboardTextByteLengthOverLimit } from './clipboard-text' +import { splitAiVaultSearchQuery } from './ai-vault-search-query-operators' import { parseWslUncPath } from './wsl-paths' import type { AiVaultAgent, @@ -174,31 +175,23 @@ export function agentLabel(agent: AiVaultAgent): string { return aiVaultAgentLabel(agent) } +/** + * The panel reading of the one shared operator split. The scanner preserves case + * so the index can treat an absolute `path:` as an identity claim; the fold is + * panel-local and safe because every comparison in `matchesQuery` is a + * case-insensitive substring probe. Removing it silently breaks all three keys. + */ export function parseVaultQuery(query: string): ParsedQuery { - const terms: string[] = [] - const repoTerms: string[] = [] - const pathTerms: string[] = [] - - for (const rawToken of tokenizeQuery(query)) { - const token = rawToken.toLowerCase() - if (token.startsWith('repo:')) { - const value = token.slice('repo:'.length) - if (value) { - repoTerms.push(value) - } - continue - } - if (token.startsWith('path:')) { - const value = token.slice('path:'.length) - if (value) { - pathTerms.push(value) - } - continue - } - terms.push(token) + const split = splitAiVaultSearchQuery(query) + return { + terms: fold(split.terms), + repoTerms: fold(split.repoTerms), + pathTerms: fold(split.pathTerms) } +} - return { terms, repoTerms, pathTerms } +function fold(values: readonly string[]): string[] { + return values.map((value) => value.trim().toLowerCase()).filter(Boolean) } function matchesQuery( @@ -290,25 +283,3 @@ function isAiVaultSessionInWorkspacePath(workspacePath: string, sessionCwd: stri // active worktree as a Windows UNC path. return isPathInsideOrEqual(workspaceWslPath.linuxPath, sessionCwd) } - -function tokenizeQuery(query: string): string[] { - const tokens: string[] = [] - // Why: keep quoted operator values (repo:/path:) intact so labels and paths - // containing spaces still match — e.g. path:"/Users/ada/My Project". - const pattern = /(repo|path):"([^"]+)"|(repo|path):'([^']+)'|"([^"]+)"|'([^']+)'|(\S+)/gi - let match: RegExpExecArray | null - while ((match = pattern.exec(query)) !== null) { - const operator = match[1] ?? match[3] - const operatorValue = match[2] ?? match[4] - if (operator && operatorValue?.trim()) { - tokens.push(`${operator.toLowerCase()}:${operatorValue.trim()}`) - continue - } - - const token = match[5] ?? match[6] ?? match[7] - if (token?.trim()) { - tokens.push(token.trim()) - } - } - return tokens -} diff --git a/src/shared/cli-argument-boundary.ts b/src/shared/cli-argument-boundary.ts index 483c75fd4f5..e27c3bcb251 100644 --- a/src/shared/cli-argument-boundary.ts +++ b/src/shared/cli-argument-boundary.ts @@ -25,7 +25,6 @@ export const CLI_BOOLEAN_FLAGS = new Set([ 'me', 'mobile', 'mobile-pairing', - 'newest', 'no-pairing', 'screen', 'parent-current', @@ -51,14 +50,15 @@ export const CLI_BOOLEAN_FLAGS = new Set([ function commandPathStartsAt( argv: readonly string[], tokenIndex: number, - path: readonly string[] + path: readonly string[], + booleanFlags: ReadonlySet ): boolean { let cursor = tokenIndex for (const part of path) { while (argv[cursor]?.startsWith('--')) { const assignment = argv[cursor].slice(2) const flag = assignment.split('=', 1)[0] - cursor += assignment.includes('=') || CLI_BOOLEAN_FLAGS.has(flag) ? 1 : 2 + cursor += assignment.includes('=') || booleanFlags.has(flag) ? 1 : 2 } if (argv[cursor] !== part) { return false @@ -71,10 +71,14 @@ function commandPathStartsAt( export function findCliCommandIndex( argv: readonly string[], commandPaths: readonly (readonly string[])[], - knownValueFlags: readonly string[] = [] + knownValueFlags: readonly string[] = [], + // Why: per-command value-less flags decide where a command path begins, so the + // boundary and the parser must read one set. Callers that know the resolved + // specs widen this; the rest get the global set. + booleanFlags: ReadonlySet = CLI_BOOLEAN_FLAGS ): number { const startsCommandAt = (index: number): boolean => - commandPaths.some((path) => commandPathStartsAt(argv, index, path)) + commandPaths.some((path) => commandPathStartsAt(argv, index, path, booleanFlags)) for (let index = 0; index < argv.length;) { const token = argv[index] @@ -87,7 +91,7 @@ export function findCliCommandIndex( const next = argv[index + 1] const takesNext = !assignment.includes('=') && - !CLI_BOOLEAN_FLAGS.has(flag) && + !booleanFlags.has(flag) && next !== undefined && !next.startsWith('--') && (knownValueFlags.includes(flag) || @@ -97,3 +101,22 @@ export function findCliCommandIndex( } return -1 } + +/** The longest registered command path that begins at `index`, or null. */ +export function findCliCommandPathAt( + argv: readonly string[], + commandPaths: readonly (readonly string[])[], + index: number, + booleanFlags: ReadonlySet = CLI_BOOLEAN_FLAGS +): readonly string[] | null { + let longest: readonly string[] | null = null + for (const path of commandPaths) { + if ( + commandPathStartsAt(argv, index, path, booleanFlags) && + (!longest || path.length > longest.length) + ) { + longest = path + } + } + return longest +} diff --git a/src/shared/protocol-version.ts b/src/shared/protocol-version.ts index df4dfde9193..e1cf7594034 100644 --- a/src/shared/protocol-version.ts +++ b/src/shared/protocol-version.ts @@ -75,7 +75,6 @@ export const JIRA_USER_FIELDS_UPDATE_REQUIRED_MESSAGE = // conditional like browser.headless.v1. export const AI_VAULT_RUNTIME_CAPABILITY = 'aiVault.v1' as const export const AI_VAULT_SESSION_TITLES_RUNTIME_CAPABILITY = 'aiVault.session-titles.v1' as const -export const AI_VAULT_SESSION_SEARCH_RUNTIME_CAPABILITY = 'aiVault.session-search.v1' as const // Why: signals a host owns browser pages with no renderer (headless serve via the // offscreen backend). Advertised only when that backend is actually available, so // clients never fall back to a local desktop browser tab for a remote-owned page. @@ -245,7 +244,6 @@ export const RUNTIME_CAPABILITIES = [ JIRA_USER_FIELDS_RUNTIME_CAPABILITY, AI_VAULT_RUNTIME_CAPABILITY, AI_VAULT_SESSION_TITLES_RUNTIME_CAPABILITY, - AI_VAULT_SESSION_SEARCH_RUNTIME_CAPABILITY, TERMINAL_QUERY_REPLY_INPUT_RUNTIME_CAPABILITY, TERMINAL_PROMPT_DELIVERY_RUNTIME_CAPABILITY, TERMINAL_PAIRED_PARKING_RUNTIME_CAPABILITY, diff --git a/src/shared/remote-runtime-request-socket.ts b/src/shared/remote-runtime-request-socket.ts index 34034f10ed0..5833a299d54 100644 --- a/src/shared/remote-runtime-request-socket.ts +++ b/src/shared/remote-runtime-request-socket.ts @@ -184,10 +184,7 @@ export async function sendRemoteRuntimeRequestOnSocket( try { ws = new WebSocket(pairing.endpoint, { - maxPayload: - method.startsWith('aiVault.') && /search/i.test(method) - ? 4 * 1024 * 1024 - : REMOTE_RUNTIME_MAX_WEBSOCKET_FRAME_BYTES + maxPayload: REMOTE_RUNTIME_MAX_WEBSOCKET_FRAME_BYTES }) } catch (error) { const message = error instanceof Error ? error.message : String(error)