From 37603a6cf3d1a022c8e940b971680b92bc26e459 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Thu, 10 Sep 2026 22:15:46 -0700 Subject: [PATCH 1/4] refactor(shared): derive WellKnownAgentType from TuiAgent (#19645) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The hand-written 22-member WellKnownAgentType union was a stale copy of the launchable-agent list, 15 members behind TuiAgent: aug, autohand, claude-agent-teams, cline, codebuff, continue, crush, goose, kilo, kimi, kiro, mistral-vibe, openclaw, qwen-code, rovo. All 21 non-'unknown' members it did carry were already TuiAgent members, so the union is now derived (`TuiAgent | 'unknown'`) and cannot drift again. 'unknown' stays outside TuiAgent: it is the "no agent identified yet" sentinel, not a launchable agent. AgentType is structurally unchanged. `(string & {})` absorbs the union, so AgentType was and remains `string` — this is a documentation/staleness fix with no behaviour change and zero consumers affected (WellKnownAgentType had none repo-wide beyond the AgentType alias itself). tui-agent.ts is a pure type union with no imports, so this adds no cycle. Adds a type-level coverage test that fails to compile if anyone reverts to a hand-written list. --- src/shared/agent-status-types.test.ts | 31 +++++++++++++++++++++++++++ src/shared/agent-status-types.ts | 28 ++++-------------------- 2 files changed, 35 insertions(+), 24 deletions(-) diff --git a/src/shared/agent-status-types.test.ts b/src/shared/agent-status-types.test.ts index b075f87ca86..f79f3c974f3 100644 --- a/src/shared/agent-status-types.test.ts +++ b/src/shared/agent-status-types.test.ts @@ -15,6 +15,8 @@ import { AGENT_STATUS_STATES, AGENT_TYPE_MAX_LENGTH } from './agent-status-types' +import type { AgentType, WellKnownAgentType } from './agent-status-types' +import type { TuiAgent } from './tui-agent' afterEach(() => { vi.restoreAllMocks() @@ -676,3 +678,32 @@ describe('normalizeAgentStatusPayload matches the JSON round trip', () => { } }) }) + +describe('WellKnownAgentType', () => { + // Compile-time proof the union is derived from TuiAgent rather than hand-copied: + // a literal list that misses any launchable agent id fails to typecheck here. + const widenTuiAgent = (agent: TuiAgent): WellKnownAgentType => agent + + it('covers every TuiAgent id plus the unknown sentinel', () => { + // ids the previous 22-member hand-written union had drifted past + const formerlyMissing: WellKnownAgentType[] = [ + 'qwen-code', + 'mistral-vibe', + 'claude-agent-teams' + ] + const sentinel: WellKnownAgentType = 'unknown' + + expect([...formerlyMissing, sentinel, widenTuiAgent('rovo')]).toEqual([ + 'qwen-code', + 'mistral-vibe', + 'claude-agent-teams', + 'unknown', + 'rovo' + ]) + }) + + it('keeps AgentType open to custom agent names', () => { + const custom: AgentType = 'some-in-house-agent' + expect(custom).toBe('some-in-house-agent') + }) +}) diff --git a/src/shared/agent-status-types.ts b/src/shared/agent-status-types.ts index 666cedb5748..3a062f7069d 100644 --- a/src/shared/agent-status-types.ts +++ b/src/shared/agent-status-types.ts @@ -5,6 +5,7 @@ import type { AgentProviderSessionMetadata } from './agent-session-resume' import type { OrchestrationFleetAttention } from './orchestration-fleet-attention' import type { AgentStatusRowFacets } from './agent-status-observation' +import type { TuiAgent } from './tui-agent' import { normalizeInteractivePromptField, normalizeOptionalField, @@ -26,30 +27,9 @@ export const AGENT_STATUS_STATES = ['working', 'blocked', 'waiting', 'done'] as export type AgentStatusState = (typeof AGENT_STATUS_STATES)[number] export type AgentWorkingMode = 'monitoring' // Why: agent types aren't a fixed set (custom agents exist); any non-empty string is -// accepted — these well-known names are just a convenience union for pattern-matching. -export type WellKnownAgentType = - | 'claude' - | 'openclaude' - | 'codex' - | 'gemini' - | 'antigravity' - | 'amp' - | 'opencode' - | 'mimo-code' - | 'cursor' - | 'copilot' - | 'aider' - | 'pi' - | 'omp' - | 'prime-agent' - | 'droid' - | 'command-code' - | 'grok' - | 'hermes' - | 'devin' - | 'ante' - | 'trae' - | 'unknown' +// accepted — the well-known names are the launchable TuiAgent ids plus the 'unknown' +// sentinel (no agent identified yet), a convenience union for pattern-matching. +export type WellKnownAgentType = TuiAgent | 'unknown' export type AgentType = WellKnownAgentType | (string & {}) /** A snapshot of a previous agent state, used to render activity blocks. From f31bfa8fb79ac19386a2a3d6d1a3578e1e6c1f7d Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 11 Sep 2026 01:24:08 -0400 Subject: [PATCH 2/4] Revert "feat(ai-vault-search): session search query engine over the index (#19750)" (#20023) This reverts commit cf7408a37fe63af4543ea6a4ece9474a993d783f. --- .gitignore | 1 - .../scripts/session-search-query-benchmark.ts | 192 ------- .../scripts/session-search-scope-benchmark.ts | 205 -------- .../session-search-tool-heavy-corpus.ts | 152 ------ .../agent-session-search-query-tuning.md | 218 -------- .../session-search-engine-test-fixture.ts | 113 ----- .../session-search-engine-types.ts | 157 ------ .../session-search-engine.test.ts | 471 ------------------ .../ai-vault-search/session-search-engine.ts | 349 ------------- .../session-search-fts5-contract.test.ts | 172 ------- .../session-search-hit-ranking.test.ts | 102 ---- .../session-search-hit-ranking.ts | 109 ---- .../session-search-index-generation.test.ts | 320 ------------ .../session-search-index-generation.ts | 97 ---- .../session-search-orphan-rows.test.ts | 186 ------- .../session-search-page-cursor.ts | 108 ---- .../session-search-paging.test.ts | 309 ------------ .../session-search-query-log.test.ts | 61 --- .../session-search-query-log.ts | 30 -- .../session-search-query-planner.test.ts | 84 ---- .../session-search-query-planner.ts | 140 ------ .../session-search-query-schema.ts | 79 --- .../session-search-retrieval.ts | 251 ---------- .../session-search-row-filter.test.ts | 137 ----- .../session-search-row-filter.ts | 90 ---- .../session-search-sidebar-parity.test.ts | 137 ----- .../session-search-snippet-marks.test.ts | 112 ----- .../ai-vault-search/session-search-snippet.ts | 126 ----- .../session-search-source-presence.ts | 40 -- .../session-search-typo-policy.test.ts | 80 --- .../session-search-typo-repair.ts | 163 ------ .../session-search-typo-scope.test.ts | 71 --- .../ai-vault-search-query-operators.test.ts | 119 ----- src/shared/ai-vault-search-query-operators.ts | 90 ---- src/shared/ai-vault-session-filters.ts | 128 +++-- 35 files changed, 63 insertions(+), 5136 deletions(-) delete mode 100644 config/scripts/session-search-query-benchmark.ts delete mode 100644 config/scripts/session-search-scope-benchmark.ts delete mode 100644 config/scripts/session-search-tool-heavy-corpus.ts delete mode 100644 docs/reference/agent-session-search-query-tuning.md delete mode 100644 src/main/ai-vault-search/session-search-engine-test-fixture.ts delete mode 100644 src/main/ai-vault-search/session-search-engine-types.ts delete mode 100644 src/main/ai-vault-search/session-search-engine.test.ts delete mode 100644 src/main/ai-vault-search/session-search-engine.ts delete mode 100644 src/main/ai-vault-search/session-search-fts5-contract.test.ts delete mode 100644 src/main/ai-vault-search/session-search-hit-ranking.test.ts delete mode 100644 src/main/ai-vault-search/session-search-hit-ranking.ts delete mode 100644 src/main/ai-vault-search/session-search-index-generation.test.ts delete mode 100644 src/main/ai-vault-search/session-search-index-generation.ts delete mode 100644 src/main/ai-vault-search/session-search-orphan-rows.test.ts delete mode 100644 src/main/ai-vault-search/session-search-page-cursor.ts delete mode 100644 src/main/ai-vault-search/session-search-paging.test.ts delete mode 100644 src/main/ai-vault-search/session-search-query-log.test.ts delete mode 100644 src/main/ai-vault-search/session-search-query-log.ts delete mode 100644 src/main/ai-vault-search/session-search-query-planner.test.ts delete mode 100644 src/main/ai-vault-search/session-search-query-planner.ts delete mode 100644 src/main/ai-vault-search/session-search-query-schema.ts delete mode 100644 src/main/ai-vault-search/session-search-retrieval.ts delete mode 100644 src/main/ai-vault-search/session-search-row-filter.test.ts delete mode 100644 src/main/ai-vault-search/session-search-row-filter.ts delete mode 100644 src/main/ai-vault-search/session-search-sidebar-parity.test.ts delete mode 100644 src/main/ai-vault-search/session-search-snippet-marks.test.ts delete mode 100644 src/main/ai-vault-search/session-search-snippet.ts delete mode 100644 src/main/ai-vault-search/session-search-source-presence.ts delete mode 100644 src/main/ai-vault-search/session-search-typo-policy.test.ts delete mode 100644 src/main/ai-vault-search/session-search-typo-repair.ts delete mode 100644 src/main/ai-vault-search/session-search-typo-scope.test.ts delete mode 100644 src/shared/ai-vault-search-query-operators.test.ts delete mode 100644 src/shared/ai-vault-search-query-operators.ts diff --git a/.gitignore b/.gitignore index 5bb1fedcaee..913dfc4a045 100644 --- a/.gitignore +++ b/.gitignore @@ -103,7 +103,6 @@ docs/** !docs/agent-skill-sharing-implementation-checklist.md !docs/mobile-terminal-shortcut-bar.md !docs/reference/ -!docs/reference/agent-session-search-query-tuning.md !docs/reference/agent-status-store.md !docs/reference/git-compatibility.md !docs/reference/headless-linux-server.md diff --git a/config/scripts/session-search-query-benchmark.ts b/config/scripts/session-search-query-benchmark.ts deleted file mode 100644 index 1471a0a17bd..00000000000 --- a/config/scripts/session-search-query-benchmark.ts +++ /dev/null @@ -1,192 +0,0 @@ -import { rm, writeFile } from 'node:fs/promises' -import { join } from 'node:path' -import { - createSessionParseStats, - parseAgentSessionFileCached, - resetSessionParseCacheForTests -} from '../../src/main/ai-vault/session-scanner-parse-cache' -import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers' -import { SessionSearchEngine } from '../../src/main/ai-vault-search/session-search-engine' -import type { - SessionSearchRequest, - SessionSearchScope -} from '../../src/main/ai-vault-search/session-search-engine-types' -import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer' -import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' -import type SyncDatabase from '../../src/main/sqlite/sync-database' -import { - writeSyntheticTranscriptCorpus, - type SyntheticCorpus, - type SyntheticCorpusOptions -} from '../../src/main/ai-vault-search/session-search-synthetic-corpus' -import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures' - -// What a query costs, and what the session candidate limit buys. Everything -// runs through the real store and the real engine over a synthetic corpus; -// never point this at a real transcript tree. - -const WARMUP = 5 -const SAMPLES = 25 - -// One query per rung the ladder can take, plus the two shapes that skip it. -const QUERIES: { name: string; request: SessionSearchRequest }[] = [ - { name: 'phrase', request: { query: '"terminal reattach"' } }, - { name: 'identifier', request: { query: 'resolveTerminalPath' } }, - { name: 'path', request: { query: 'src/main/ai-vault/session-transcript-reader.ts' } }, - { name: 'prose', request: { query: 'why is the daemon snapshot stale' } }, - { name: 'typo', request: { query: 'reattahc worktre' } }, - { name: 'common-term', request: { query: 'index' } }, - { name: 'operator-only', request: { query: 'repo:app-3' } }, - { name: 'scoped', request: { query: 'worktree', filters: { scopePaths: ['/repo/app-3'] } } } -] - -type Timing = { p50: number; p95: number } - -function percentile(sorted: readonly number[], fraction: number): number { - const at = Math.min(sorted.length - 1, Math.floor(sorted.length * fraction)) - return Math.round((sorted[at] ?? 0) * 100) / 100 -} - -function timing(samples: number[]): Timing { - const sorted = [...samples].sort((left, right) => left - right) - return { p50: percentile(sorted, 0.5), p95: percentile(sorted, 0.95) } -} - -function time(engine: SessionSearchEngine, request: SessionSearchRequest): number { - const started = performance.now() - engine.search(request) - return performance.now() - started -} - -async function indexCorpus( - options: SyntheticCorpusOptions -): Promise<{ corpus: SyntheticCorpus; db: SyncDatabase; release: () => void }> { - resetSessionParseCacheForTests() - const corpus = await writeSyntheticTranscriptCorpus(options) - const store = new SessionSearchStore(join(corpus.root, 'index.sqlite'), (error) => { - throw error - }) - const unregister = registerSessionSearchIndexConsumer(store) - const stats = createSessionParseStats() - for (const path of corpus.files) { - await parseAgentSessionFileCached( - await sessionCandidate('claude', path), - process.platform, - stats - ) - } - return { - corpus, - // The handle a composed reader gets. Every read here is one synchronous - // statement, which is the contract that comes with it. - db: store.connection, - release: () => { - unregister() - resetTranscriptConsumersForTests() - resetSessionParseCacheForTests() - store.close() - } - } -} - -/** Per-query and overall latency for one scope. */ -function scopeReport(db: SyncDatabase, scope: SessionSearchScope): Record { - const engine = new SessionSearchEngine(db) - const everything: number[] = [] - const perQuery: Record = {} - for (const { name, request } of QUERIES) { - const scoped = { ...request, scope } - for (let run = 0; run < WARMUP; run++) { - engine.search(scoped) - } - const samples = Array.from({ length: SAMPLES }, () => time(engine, scoped)) - everything.push(...samples) - const result = engine.search(scoped) - perQuery[name] = { ...timing(samples), hits: result.hits.length, route: result.planner.route } - } - return { ...timing(everything), perQuery } -} - -/** - * The candidate limit only costs anything once there are more matching sessions - * than the limit, so this runs over many short sessions rather than the wide - * corpus above. Limits are interleaved sample by sample: run back to back, the - * first configuration pays for every page the OS cache had not seen yet and the - * ordering alone moves p95 by more than the limit does. - */ -function candidateSweep(db: SyncDatabase, limits: readonly number[]): Record { - const request: SessionSearchRequest = { query: 'index', limit: 20 } - const engines = new Map( - limits.map((limit) => [limit, new SessionSearchEngine(db, { sessionCandidateLimit: limit })]) - ) - const samples = new Map(limits.map((limit) => [limit, [] as number[]])) - for (let run = 0; run < WARMUP; run++) { - for (const engine of engines.values()) { - engine.search(request) - } - } - for (let run = 0; run < SAMPLES; run++) { - for (const limit of limits) { - samples.get(limit)!.push(time(engines.get(limit)!, request)) - } - } - const report: Record = {} - for (const limit of limits) { - const result = engines.get(limit)!.search(request) - report[String(limit)] = { - ...timing(samples.get(limit)!), - truncated: result.truncated.candidates, - // Pages a caller could walk before the limit stops handing out sessions. - reachablePages: Math.ceil(limit / (request.limit ?? 20)) - } - } - return report -} - -const wide = await indexCorpus({ sessions: Number(process.env.SESSIONS ?? 40) }) -let report: string -try { - const scope = { - all: scopeReport(wide.db, 'all'), - conversation: scopeReport(wide.db, 'conversation') - } - wide.release() - await rm(wide.corpus.root, { recursive: true, force: true }) - - // Many short sessions: what makes the candidate limit binding is the session - // count, not the byte count. - const many = await indexCorpus({ sessions: 2500, turnsPerSession: 1, seed: 7 }) - try { - report = JSON.stringify( - { - scopeCorpus: { - sessions: wide.corpus.files.length, - transcriptMb: Math.round((wide.corpus.transcriptBytes / 1024 / 1024) * 100) / 100, - messages: wide.corpus.messageCount - }, - scope, - candidateCorpus: { - sessions: many.corpus.files.length, - transcriptMb: Math.round((many.corpus.transcriptBytes / 1024 / 1024) * 100) / 100 - }, - candidateSweep: candidateSweep(many.db, [200, 600, 1200, 2400]) - }, - null, - 2 - ) - } finally { - many.release() - await rm(many.corpus.root, { recursive: true, force: true }) - } -} catch (error) { - await rm(wide.corpus.root, { recursive: true, force: true }) - throw error -} - -// Why a file as well as stdout: a runner that intercepts console output -// (vitest does) would otherwise swallow the whole report. -const out = process.env.BENCH_OUT -if (out) { - await writeFile(out, `${report}\n`) -} -console.log(report) diff --git a/config/scripts/session-search-scope-benchmark.ts b/config/scripts/session-search-scope-benchmark.ts deleted file mode 100644 index 306387cbdda..00000000000 --- a/config/scripts/session-search-scope-benchmark.ts +++ /dev/null @@ -1,205 +0,0 @@ -import { rm, writeFile } from 'node:fs/promises' -import { join } from 'node:path' -import { - createSessionParseStats, - parseAgentSessionFileCached, - resetSessionParseCacheForTests -} from '../../src/main/ai-vault/session-scanner-parse-cache' -import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers' -import { SessionSearchEngine } from '../../src/main/ai-vault-search/session-search-engine' -import type { - SessionSearchRequest, - SessionSearchScope -} from '../../src/main/ai-vault-search/session-search-engine-types' -import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer' -import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' -import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures' -import type SyncDatabase from '../../src/main/sqlite/sync-database' -import { writeToolHeavyCorpus, type ToolHeavyCorpus } from './session-search-tool-heavy-corpus' - -// What each scope costs on an index the size of a real transcript tree. -// -// The 10.5 MB corpus in `session-search-query-benchmark.ts` sizes the route -// ladder; this one sizes the corpus. `conversation` is a column filter over the -// one FTS table rather than a second table of its own, and the whole cost of -// that decision is how much of `messages_fts` a conversation query has to read -// past — which is set by how much of a transcript is tool output. -// -// Synthetic, always: this must never be pointed at a real transcript. - -const WARMUP = 5 - -/** Conversation-shaped queries; every term is one the prose actually uses. */ -const QUERIES = [ - 'terminal reattach', - 'stale snapshot', - 'daemon cursor', - 'worktree index', - 'publish transaction', - 'relay daemon', - 'session cursor', - 'because stale', - 'terminal worktree', - 'index snapshot', - 'reattach cursor', - 'transaction relay', - 'snapshot session', - 'daemon publish', - 'worktree terminal', - 'cursor index', - 'stale relay', - 'session transaction', - 'publish snapshot', - 'reattach daemon' -] - -async function indexCorpus( - corpus: ToolHeavyCorpus -): Promise<{ db: SyncDatabase; release: () => void }> { - resetSessionParseCacheForTests() - const store = new SessionSearchStore(join(corpus.root, 'index.sqlite'), (error) => { - throw error - }) - const unregister = registerSessionSearchIndexConsumer(store) - const stats = createSessionParseStats() - for (const path of corpus.files) { - await parseAgentSessionFileCached( - await sessionCandidate('claude', path), - process.platform, - stats - ) - } - return { - // The store's own handle, which is what a composed reader gets: every - // retrieval is one synchronous statement, so nothing pins a WAL snapshot. - db: store.connection, - release: () => { - unregister() - resetTranscriptConsumersForTests() - resetSessionParseCacheForTests() - store.close() - } - } -} - -type Timing = { p50: number; p95: number } - -function timing(samples: readonly number[]): Timing { - const sorted = [...samples].sort((left, right) => left - right) - const at = (fraction: number): number => { - const index = Math.min(sorted.length - 1, Math.floor(sorted.length * fraction)) - return Math.round((sorted[index] ?? 0) * 100) / 100 - } - return { p50: at(0.5), p95: at(0.95) } -} - -/** - * The query sets, one per rung of the ladder the engine may take. - * - * Which rung each one reaches is not forced, it is observed: samples are - * bucketed by the route the engine reports, so the table says what was measured - * rather than what was intended, and a query that lands on a different rung - * than expected shows up as a bucket rather than as a wrong number. - */ -function queries(): string[] { - const run = (index: number, length: number): string => - Array.from({ length }, (_unused, step) => QUERIES[(index + step) % QUERIES.length]).join(' ') - return [ - // Two terms, unquoted: not literal, so straight to OR. - ...QUERIES, - // Two terms, quoted: literal, and on this corpus any two of fourteen words - // sit next to each other somewhere, so the phrase rung answers. - ...QUERIES.map((query) => `"${query}"`), - // Eight terms, quoted: an ordered run that long does not occur in 105 MB of - // draws from fourteen words, so the phrase rung misses and AND answers. - ...QUERIES.map((_query, index) => `"${run(index, 4)}"`) - ] -} - -type Bucket = { samples: number[]; hits: number } - -/** - * Both scopes over the same queries, interleaved scope by scope: run back to - * back, the first one pays for every page the OS cache had not seen and the - * ordering moves p95 more than the scope does. - */ -function scopeReport(db: SyncDatabase): Record { - const engine = new SessionSearchEngine(db) - const scopes: SessionSearchScope[] = ['all', 'conversation'] - const requests: SessionSearchRequest[] = queries().map((query) => ({ query })) - const buckets = new Map() - for (let run = 0; run < WARMUP; run++) { - for (const scope of scopes) { - for (const request of requests) { - engine.search({ ...request, scope }) - } - } - } - for (const request of requests) { - for (const scope of scopes) { - const started = performance.now() - const result = engine.search({ ...request, scope }) - const elapsed = performance.now() - started - const key = `${result.planner.route}/${scope}` - const bucket = buckets.get(key) ?? { samples: [], hits: 0 } - bucket.samples.push(elapsed) - bucket.hits += result.hits.length - buckets.set(key, bucket) - } - } - const report: Record = {} - for (const [key, bucket] of [...buckets].sort(([left], [right]) => left.localeCompare(right))) { - report[key] = { ...timing(bucket.samples), samples: bucket.samples.length, hits: bucket.hits } - } - return report -} - -/** Bytes the FTS table occupies, which is the cost the deleted second table saved. */ -function indexBytes(db: SyncDatabase): Record | { unavailable: string } { - try { - const sum = (where: string, ...values: string[]): number => - Number( - ( - db - .prepare(`SELECT COALESCE(SUM(pgsize),0) AS bytes FROM dbstat ${where}`) - .get(...values) as { bytes: number } - ).bytes - ) - return { total: sum(''), messagesFts: sum('WHERE name LIKE ?', 'messages_fts%') } - } catch { - // dbstat is a compile-time option; the latency numbers stand without it. - return { unavailable: 'no dbstat' } - } -} - -const corpus = await writeToolHeavyCorpus({ - targetBytes: Number(process.env.CORPUS_MB ?? 100) * 1024 * 1024, - toolShare: Number(process.env.TOOL_SHARE ?? 0.9) -}) -let report: string -const indexed = await indexCorpus(corpus) -try { - report = JSON.stringify( - { - corpus: { - sessions: corpus.files.length, - transcriptMb: Math.round((corpus.transcriptBytes / 1024 / 1024) * 100) / 100, - toolShareOfMessageText: - Math.round((corpus.toolBytes / (corpus.toolBytes + corpus.proseBytes)) * 1000) / 1000 - }, - indexBytes: indexBytes(indexed.db), - route: scopeReport(indexed.db) - }, - null, - 2 - ) -} finally { - indexed.release() - await rm(corpus.root, { recursive: true, force: true }) -} - -const out = process.env.BENCH_OUT -if (out) { - await writeFile(out, `${report}\n`) -} -console.log(report) diff --git a/config/scripts/session-search-tool-heavy-corpus.ts b/config/scripts/session-search-tool-heavy-corpus.ts deleted file mode 100644 index 050535a00cf..00000000000 --- a/config/scripts/session-search-tool-heavy-corpus.ts +++ /dev/null @@ -1,152 +0,0 @@ -import { mkdtemp, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' - -// The corpus the scope benchmark runs over. Written here rather than by -// `session-search-synthetic-corpus.ts` because what it costs to answer a -// conversation query out of the one FTS table turns on the property that -// generator fixes: how much of a transcript is tool output. -// -// Synthetic, always. This must never be pointed at a real transcript. - -const PROSE = [ - 'terminal', - 'reattach', - 'worktree', - 'the', - 'index', - 'cursor', - 'publish', - 'transaction', - 'relay', - 'daemon', - 'snapshot', - 'because', - 'stale', - 'session' -] -// Tool output is paths, hashes and log lines — and the same words the -// conversation uses, because a `rg` over this repository prints them. That -// overlap is what the benchmark turns on: it is what makes a conversation -// term's posting list carry rows the column filter then has to discard. A tool -// vocabulary disjoint from the prose would leave nothing to discard and measure -// the wrong thing. -const TOOL_ONLY = [ - 'src/main/ai-vault/session-transcript-reader.ts', - 'node_modules/.pnpm/typescript@5.9.2', - '0x00007ff8', - 'ENOENT', - 'drwxr-xr-x', - '2026-09-10T00:00:00.000Z', - 'sha256:9f2c1a', - 'chunk-VHQ4NWQK.js', - 'warning:', - 'resolveTerminalPath', - 'byteOffset', - 'MAX_RETRIES' -] -// Half the tool tokens are conversation words. Deliberately pessimistic: the -// more of a query term lives in `tool_text`, the more the column filter costs, -// so a number measured here holds on a real transcript tree. -const TOOL = [...PROSE, ...TOOL_ONLY] - -function mulberry32(seed: number): () => number { - let state = seed >>> 0 - return () => { - state = (state + 0x6d2b79f5) >>> 0 - let t = Math.imul(state ^ (state >>> 15), 1 | state) - t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t - return ((t ^ (t >>> 14)) >>> 0) / 4294967296 - } -} - -function words(random: () => number, vocabulary: readonly string[], count: number): string { - const out: string[] = [] - for (let index = 0; index < count; index++) { - out.push(vocabulary[Math.floor(random() * vocabulary.length)]!) - } - return out.join(' ') -} - -export type ToolHeavyCorpus = { - root: string - files: string[] - transcriptBytes: number - toolBytes: number - proseBytes: number -} - -/** - * Claude JSONL transcripts whose tool output is `toolShare` of the message text. - * One turn is a user question, an assistant answer, a tool call and its output; - * only the last one grows with the share. - */ -export async function writeToolHeavyCorpus(args: { - targetBytes: number - toolShare: number - seed?: number -}): Promise { - const random = mulberry32(args.seed ?? 11) - const root = await mkdtemp(join(tmpdir(), 'orca-search-convfts-')) - const files: string[] = [] - const proseWordsPerTurn = 160 - // Tool and prose words are not the same length, so the share is over bytes. - const proseBytesPerTurn = proseWordsPerTurn * 6 - const toolWordCount = Math.max( - 1, - Math.round((proseBytesPerTurn * args.toolShare) / (1 - args.toolShare) / 22) - ) - let transcriptBytes = 0 - let toolBytes = 0 - let proseBytes = 0 - for (let session = 0; transcriptBytes < args.targetBytes; session++) { - const sessionId = `00000000-0000-4000-8000-${String(session).padStart(12, '0')}` - const lines: string[] = [] - for (let turn = 0; turn < 40; turn++) { - const at = new Date(1740000000000 + turn * 60_000).toISOString() - const question = words(random, PROSE, 40) - const answer = words(random, PROSE, proseWordsPerTurn - 40) - const output = words(random, TOOL, toolWordCount) - proseBytes += Buffer.byteLength(question) + Buffer.byteLength(answer) - toolBytes += Buffer.byteLength(output) - lines.push( - JSON.stringify({ - type: 'user', - sessionId, - timestamp: at, - cwd: `/repo/app-${session % 7}`, - gitBranch: 'main', - message: { role: 'user', content: question } - }), - JSON.stringify({ - type: 'assistant', - sessionId, - timestamp: at, - message: { - role: 'assistant', - model: 'claude-fable-5', - content: [ - { type: 'text', text: answer }, - { type: 'tool_use', name: 'Bash', input: { command: 'rg needle' } } - ] - } - }), - JSON.stringify({ - type: 'user', - sessionId, - timestamp: at, - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: 'toolu_1', content: output }] - } - }) - ) - } - const path = join(root, `${sessionId}.jsonl`) - const body = `${lines.join('\n')}\n` - await writeFile(path, body) - transcriptBytes += Buffer.byteLength(body) - files.push(path) - } - return { root, files, transcriptBytes, toolBytes, proseBytes } -} diff --git a/docs/reference/agent-session-search-query-tuning.md b/docs/reference/agent-session-search-query-tuning.md deleted file mode 100644 index fcae799f8a6..00000000000 --- a/docs/reference/agent-session-search-query-tuning.md +++ /dev/null @@ -1,218 +0,0 @@ -# Agent session search: query tuning - -What a search costs, and what the knobs in `src/main/ai-vault-search/session-search-engine.ts` -buy. Every number here comes from `config/scripts/session-search-query-benchmark.ts` -over the synthetic corpus in `session-search-synthetic-corpus.ts`, except the -`conversation_fts` shoot-out, which writes its own corpus because the answer -turns on how much of a transcript is tool output. Nothing in this file was -measured against a real transcript, and neither benchmark must ever be pointed -at one. - -## Running it - -The benchmark is a top-level-await module that imports the main-process tree by -extensionless path, so it needs a bundler-backed runner rather than bare `node`: - -```sh -cat > src/main/ai-vault-search/zz-bench.test.ts <<'EOF' -import { it } from 'vitest' -it('runs', { timeout: 1_800_000 }, async () => { - await import('../../../config/scripts/session-search-query-benchmark') -}) -EOF -BENCH_OUT=/tmp/ss-query-bench.json pnpm test src/main/ai-vault-search/zz-bench.test.ts -rm src/main/ai-vault-search/zz-bench.test.ts -``` - -The `conversation_fts` shoot-out below runs the same way, importing -`config/scripts/session-search-conversation-fts-benchmark` instead, with -`CORPUS_MB` and `TOOL_SHARE` to size and shape its corpus. `config/scripts` is -not inside any typecheck project, so while that throwaway test exists `tsc` -reports TS6307 for each script it pulls in; delete it and the run is clean -again. - -`BENCH_OUT` exists because vitest intercepts `console.log`; the report is written -to that path as well as printed. - -## Scope: what the second FTS table buys a reader - -Corpus: 40 synthetic Claude transcripts, 10.5 MB, 9,600 messages, indexed through -the real store. Eight queries, one per rung of the route ladder plus the two -shapes that skip it; 5 warm-up runs and 25 samples each. Apple silicon, warm page -cache, machine otherwise idle. Milliseconds, and p95 over 25 samples moves -several milliseconds run to run if anything else is competing for the disk. - -| Scope | p50 | p95 | -| -------------- | ---- | ---- | -| `all` | 7.22 | 8.94 | -| `conversation` | 5.33 | 7.86 | - -Per query, `all` then `conversation` (p50 / p95): - -| Query | `all` | `conversation` | -| ------------------------------------------------ | ------------ | -------------- | -| `"terminal reattach"` (phrase) | 5.24 / 8.42 | 2.97 / 3.24 | -| `resolveTerminalPath` (identifier) | 7.55 / 8.94 | 6.47 / 6.72 | -| `src/main/…/session-transcript-reader.ts` (path) | 8.69 / 10.12 | 7.78 / 8.04 | -| `why is the daemon snapshot stale` (prose) | 7.84 / 8.57 | 5.90 / 7.01 | -| `reattahc worktre` (typo repair) | 7.30 / 7.39 | 5.53 / 5.89 | -| `index` (common term) | 5.45 / 5.66 | 3.81 / 4.02 | -| `repo:app-3` (operator only) | 0.12 / 0.16 | 0.10 / 0.10 | -| `worktree` scoped to one cwd | 1.47 / 1.63 | 1.25 / 1.49 | - -Reading it: - -- `conversation` is about 1.4x faster at p50 and 1.1x at p95, and it is a column - filter over the same table rather than a table of its own. Narrowing to the - two prose columns is what buys the gap: fewer postings to score. It is also - the scope where a match is something a person wrote rather than something a - tool printed. -- A `scopePaths` query is the cheapest real search on the page. It is the one - narrowing SQL can express exactly, so it seeks `sessions_cwd_key` and hands - ranking a small candidate set. -- The operator-only figure is a floor, not a typical cost. `repo:` and `path:` - are applied in JS over retrieved rows (see `session-search-row-filter` for why - they cannot be pushed into SQL), so their cost tracks how many sessions the - walk has to read before it fills a candidate set. This corpus has 40 sessions, - which is one page of that walk; an index where few sessions match the operator - will read up to the ceiling in `session-search-retrieval` instead. - -## What the conversation scope costs at real corpus size - -`conversation` was a second FTS table holding a copy of the two prose columns. -It is a column filter now — `{user_text assistant_text}: (…)` with bm25 weights -that zero the other two — and PR 2 deleted the table on the strength of the -shoot-out this section used to hold: the filter came in at 1.16-1.36x the p95 of -the dedicated table, under the 2x bar, while the table cost a tenth of the index -to maintain. What follows is what the shipped schema actually does, measured -again on the same corpus after the table went and tool rows were capped. - -Corpus: Claude transcripts from `config/scripts/session-search-tool-heavy-corpus.ts`, -105 MB, indexed through the real store, at two points in the 80-97% band a real -transcript tree sits in. Half the tokens in tool output are words the -conversation also uses, so a conversation term really does have postings the -filter must discard. Twenty queries per rung, both scopes interleaved query by -query, warm cache; `config/scripts/session-search-scope-benchmark.ts`, run twice. - -| Tool share | Rung | `all` p50 / p95 | `conversation` p50 / p95 | -| ---------- | ------ | --------------- | ------------------------ | -| 86% | phrase | 16.69 / 17.48 | 13.08 / 13.52 | -| 86% | or | 31.91 / 35.74 | 22.25 / 23.87 | -| 86% | and | 70.04 / 74.00 | 53.47 / 59.39 | -| 93% | phrase | 9.14 / 13.36 | 7.23 / 8.51 | -| 93% | or | 16.46 / 18.70 | 12.34 / 14.88 | -| 93% | and | 39.65 / 43.44 | 31.05 / 32.92 | - -Three things to read out of it. - -**The filter is a win, not a cost.** Every rung is faster narrow than wide, by -1.2x to 1.4x at p50. The shoot-out compared the filter against a table built for -exactly this query; against the wide table it replaces, it does what the second -table did, which is read fewer postings. - -**The `and` rung is where the corpus size shows.** Those queries are eight terms, -chosen so no ordered run that long occurs and the phrase rung has to miss; a -real two-term AND sits nearer the phrase row. It is also the noisiest: the -second run's p95 reached 140 ms on one bucket, which is what twenty samples of a -70 ms query buys. Read the p50 column. - -**The index is far smaller than the shoot-out's was.** 57 MB at 93% tool output -and 103 MB at 86%, against roughly 150 MB for `messages_fts` alone before PR 2 -capped an indexed tool row at 3,072 characters. Most of a tool-heavy transcript -is now not in the index at all, which moves every number above and is the larger -effect of the two. - -What is **not** measured here is relevance, and the column filter does carry one -ranking difference the deleted table did not. FTS5's bm25 normalises by the -whole row's length and has no per-column length, so two rows with identical -prose score differently when one also holds tool output. The rowid set is -unchanged, which is what the deletion was decided on; the order within it can -move. `session-search-engine.test.ts` pins the direction. - -## `sessionCandidateLimit` - -The reviewer's F13: this is a tunable default, not a constant. It bounds how many -sessions the SQL hands ranking, so it bounds both retrieval cost and how deep a -caller can page before the answer simply stops. - -The limit only costs anything once more sessions match than the limit allows, so -this is measured over a second corpus: 2,500 one-turn transcripts, 10.9 MB, every -one of them matching the query. Limits are interleaved sample by sample, because -run back to back the first configuration pays for every page the OS cache had not -seen and the ordering alone moves p95 further than the limit does. - -| Limit | p50 | p95 | Pages of 20 a caller can reach | -| ----- | ----- | ----- | ------------------------------ | -| 200 | 6.85 | 7.21 | 10 | -| 600 | 7.93 | 8.36 | 30 | -| 1200 | 9.55 | 10.53 | 60 | -| 2400 | 12.32 | 13.45 | 120 | - -600 is the default: it costs about 16% over 200 at p50 and buys three times the -reachable depth, and the curve only turns steep past 1200. A host with a much -larger index can raise it; the result's `truncated.candidates` says when the limit -was the thing that cut the answer, so a caller never has to guess. - -What is **not** measured here is relevance. These numbers say what a limit costs, -not what it retrieves. The MRR figures quoted in the BM25 weights -(`session-search-retrieval.ts`) and in the identifier shadow column -(`session-search-identifier-split.ts`) come from the original retrieval shoot-out -on real transcripts and are not reproducible from this repository. Any change to -the limit justified on relevance grounds needs an eval set, not this benchmark. - -## What typo repair costs - -The repair is the one rung whose cost tracks the size of the vocabulary rather -than the size of a result. It only runs for a term the scope has no posting for, -so an ordinary query never pays it; a query of nonsense pays it once per term. - -Measured over a synthetic vocabulary of 1.6 M distinct terms, every term in two -rows so none is filtered out: - -| Query | p50 | -| -------------------------------------- | ------ | -| one known term (no repair) | 11 ms | -| one unknown term | 10 ms | -| 39 unknown 12-character terms (480 ch) | 387 ms | -| 12 unknown 40-character terms | 99 ms | - -Two things follow. The cost is linear in unknown terms and in vocabulary size, -and `search` is synchronous, so a 512-character query of nonsense holds the -thread for a third of a second on an index that large. And the scoped-count fix -made this cheaper rather than dearer — it was 737 ms before — because ordering -the vocabulary scan by term drops the sort that ordering by `doc` required, and -the counts it added are at most eight bounded probes per prefix. A cap on -unknown terms per query is recorded as a follow-up in the split plan. - -## Page warmup, dropped - -PR 2 deferred `warm()` — a sliced read of `messages` that pulls its pages into -the OS cache before the first query — to whoever knew which pages a read -touches. It is not re-added here, for two reasons. The measurement that -justified it (first query 1.3 s to 0.45 s) was on a 4 GB index, and neither -corpus in this file is within an order of magnitude of that, so PR 4 cannot -show a win: removing the call moved the 10.5 MB corpus's p50 by less than the -run-to-run spread. And it is a cancellable background pass, which needs an owner -with a lifecycle; a query library that holds no timers has nothing to hang the -`stopped()` on, and a fire-and-forget async read from a synchronous `search` is -a rejection nothing can supervise. It belongs with the indexer in PR 3b, which -already owns starting and stopping work. - -## Not settled here - -Which process may open, unlink and rebuild the index is PR 3b's decision. A -second handle that finds an older schema version replaces the file while a live -store keeps answering from the unlinked inode, and this PR is what first makes -that reachable, because it is the first thing that reads. What PR 4 does is -refuse to make it worse. The engine carries its own schema — the vocabulary, the -query log and the generation triggers — and re-creates whatever of it is missing -on every search, so a dropped object heals rather than degrading. - -The one it cannot re-create is the vocabulary's source, because `messages_fts` -is the store's. With one FTS table that is also the end of the degrade: there is -no second corpus to answer from, so an engine over an index mid-rebuild names -typo repair as unavailable and then fails on the table it cannot read, which is -the honest outcome — an empty page would read as an answer. `unavailable` can -therefore no longer be reported alongside a successful search, and PR 5 should -decide whether the field survives into the contract; it becomes reachable again -the day something opens the index read-only. diff --git a/src/main/ai-vault-search/session-search-engine-test-fixture.ts b/src/main/ai-vault-search/session-search-engine-test-fixture.ts deleted file mode 100644 index 694a3d6397f..00000000000 --- a/src/main/ai-vault-search/session-search-engine-test-fixture.ts +++ /dev/null @@ -1,113 +0,0 @@ -import type { TranscriptMessageRole } from '../ai-vault/session-transcript-consumers' -import type SyncDatabase from '../sqlite/sync-database' -import { SessionSearchEngine, type SessionSearchEngineOptions } from './session-search-engine' -import { cwdKey } from './session-search-file-records' -import { identifierShadowText } from './session-search-identifier-split' -import { SessionSearchStore } from './session-search-store' -import { - openSessionSearchIndexFile, - type SessionSearchIndexFile -} from './session-search-index-test-fixture' - -// Synthetic index rows for the query tests. The write path has its own tests; -// driving it here would make every retrieval assertion depend on the parser. - -export type SessionSearchHarness = { - /** The engine's own connection; the store next to it keeps a second, private one. */ - db: SyncDatabase - /** A real writer on the same file, so a test can move the index under the engine. */ - store: SessionSearchStore - engine: SessionSearchEngine - close: () => Promise -} - -export async function openSessionSearchHarness( - name: string, - options: SessionSearchEngineOptions = {} -): Promise { - const index: SessionSearchIndexFile = await openSessionSearchIndexFile(name) - const store = new SessionSearchStore(index.path, (error) => { - throw error - }) - // Constructed before any row is planted, because constructing it is what - // installs the generation triggers the planted rows have to move. - const engine = new SessionSearchEngine(index.db, options) - return { - db: index.db, - store, - engine, - close: async () => { - store.close() - await index.close() - } - } -} - -export type SyntheticSession = { - id: number - cwd?: string | null - text?: string - /** Rows of `text` to write; one session with many rows is one hit. */ - rows?: number - role?: TranscriptMessageRole - /** - * Written into `tool_text` alongside `text`, which is the one row shape the - * conversation scope has to exclude while the `all` scope keeps it. - */ - toolText?: string - agent?: string - updatedAt?: string - messageCount?: number - /** Written into `files`, which is what makes the source `present`. */ - filePath?: string | null - /** `sessions.file_path`: the transcript `path:` searches alongside cwd. */ - sessionFilePath?: string -} - -/** One session and its message rows, in both FTS tables the way the writer does. */ -export function addSyntheticSession(db: SyncDatabase, session: SyntheticSession): void { - const { - id, - cwd = '/repo/app', - text = 'needle', - rows = 1, - role = 'user', - toolText = '', - agent = 'claude', - updatedAt = `2026-09-${String((id % 28) + 1).padStart(2, '0')}T00:00:00.000Z`, - messageCount = rows, - filePath = `/synthetic/${id}.jsonl`, - sessionFilePath = `/synthetic/${id}.jsonl` - } = session - db.prepare( - `INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,updated_at,message_count,resume_command) - VALUES (?,?,?,?,'fixture',?,?,?,?,'resume')` - ).run(id, agent, String(id), sessionFilePath, cwd, cwdKey(cwd), updatedAt, messageCount) - if (filePath !== null) { - db.prepare( - 'INSERT INTO files(path,byte_offset,mtime_ms,session_row_id) VALUES (?,0,1740000000000,?)' - ).run(filePath, id) - } - for (let row = 0; row < rows; row++) { - const messageId = Number( - db - .prepare('INSERT INTO messages(session_row_id,role,ts) VALUES (?,?,?)') - .run(id, role, updatedAt).lastInsertRowid - ) - const user = role === 'user' ? text : '' - const assistant = role === 'assistant' ? text : '' - const tool = role === 'tool' ? `${text} ${toolText}`.trim() : toolText - db.prepare( - 'INSERT INTO messages_fts(rowid,user_text,assistant_text,tool_text,identifiers) VALUES (?,?,?,?,?)' - ).run(messageId, user, assistant, tool, identifierShadowText(`${text} ${toolText}`)) - } -} - -export function markFork(db: SyncDatabase, ids: readonly number[], hash: string): void { - for (const id of ids) { - db.prepare('UPDATE sessions SET content_hash = ?, content_hash_count = 8 WHERE id = ?').run( - hash, - id - ) - } -} diff --git a/src/main/ai-vault-search/session-search-engine-types.ts b/src/main/ai-vault-search/session-search-engine-types.ts deleted file mode 100644 index 861130be59b..00000000000 --- a/src/main/ai-vault-search/session-search-engine-types.ts +++ /dev/null @@ -1,157 +0,0 @@ -import type { AiVaultAgent } from '../../shared/ai-vault-types' -import type { TranscriptMessageRole } from '../ai-vault/session-transcript-consumers' -import type { SessionSearchUnavailableFeature } from './session-search-query-schema' - -// ENGINE types, deliberately not in src/shared: nothing here is a wire type. -// PR 5 owns the public contract and lifts what a caller may actually receive; -// until then a field can be added, renamed or dropped without a compat story. - -export const SESSION_SEARCH_LIMIT_DEFAULT = 20 -export const SESSION_SEARCH_LIMIT_MAX = 100 -// Longer than this is not a query, and FTS5 pays for every term it plans. -export const SESSION_SEARCH_QUERY_MAX_LENGTH = 512 - -// Snippet match markers. Why doubled: single brackets are everywhere in code -// transcripts (`arr[0]`, regex classes, markdown links) and would read as -// matches; doubled ones are rare. -export const SESSION_SEARCH_SNIPPET_MARK_OPEN = '[[' -export const SESSION_SEARCH_SNIPPET_MARK_CLOSE = ']]' - -/** - * Which corpus answers the query. - * - * - `conversation`: user and assistant turns only, as a column filter over - * `messages_fts` (see `scopedExpression`). - * - `all`: those turns plus tool calls and tool output, and the identifier - * shadow column, from `messages_fts`. - * - * The engine searches exactly the scope it is given. Switching corpus as the - * user types is a UI policy and lives in the panel (PR 7); an engine that - * second-guessed the scope would make a result impossible to reproduce from - * its own request. - */ -export type SessionSearchScope = 'conversation' | 'all' - -export type SessionSearchSort = 'relevance' | 'newest' - -export type SessionSearchFilters = { - agents?: readonly AiVaultAgent[] - /** Only sessions whose cwd is that path or inside it. */ - scopePaths?: readonly string[] - /** ISO timestamp; only sessions updated at or after it. */ - since?: string - sort?: SessionSearchSort -} - -export type SessionSearchRequest = { - query: string - /** Default `all`. */ - scope?: SessionSearchScope - limit?: number - /** From a previous response's `page.cursor`; only valid in its own generation. */ - cursor?: string - filters?: SessionSearchFilters -} - -export type SessionSearchRoute = 'phrase' | 'and' | 'or' | 'typo+phrase' | 'typo+and' | 'typo+or' - -/** - * How the query was executed. Diagnostics, not an answer: PR 5 decides which of - * these a caller ever sees (the reviewer's F5/F7 want them behind `debug`). - */ -export type SessionSearchPlannerReport = { - route: SessionSearchRoute - /** - * The whole body the repaired plan searched, in query order, when any term - * was changed. Not just the corrected terms: a caller rendering "searched - * for" needs the query it actually ran, and a repair never drops a term the - * original kept. A corrected term carries the index's own spelling, which the - * tokenizer has case-folded; untouched terms keep the case they were typed in. - */ - repairedTerms?: string[] - /** The corpus the route ran against; today always the requested scope. */ - tier: SessionSearchScope -} - -/** - * Where a source stands according to the index's own `files` table. The query - * path never stats a transcript, so it can report that the index has a live - * file record for a session or that it has none, and never that a source is - * gone: only a proven deletion may claim `missing`, and proving one is the - * indexer's job (docs/reference/ssh-execution-boundary.md). - */ -export type SessionSearchSourcePresence = 'present' | 'unverifiable' - -export type SessionSearchEvidence = { - role: TranscriptMessageRole - timestamp: string | null - /** FTS5 snippet with the matched terms wrapped in `[[` `]]`. */ - snippet: string - /** The snippet hit the engine's per-hit ceiling and was cut. */ - snippetTruncated?: boolean -} - -export type SessionSearchHit = { - agent: AiVaultAgent - sessionId: string - filePath: string - codexHome: string | null - title: string - cwd: string | null - branch: string | null - updatedAt: string | null - messageCount: number - resumeCommand: string - score: number - /** Sessions folded into this hit (forks sharing an opening prefix); absent when unique. */ - duplicateCount?: number - source: SessionSearchSourcePresence - /** Null when the operators alone put this session on the page, with no text match. */ - evidence: SessionSearchEvidence | null -} - -export type SessionSearchPage = { - /** Null when this page is the last one. */ - cursor: string | null - hasMore: boolean -} - -export type SessionSearchTruncation = { - /** - * Ranking saw only the first `sessionCandidateLimit` sessions, so a session - * past that cut cannot appear on any page of this query. - */ - candidates: boolean - /** Hits on this page whose snippet was cut. */ - snippets: number - /** - * The query itself was cut before it was searched: past the length ceiling, - * or past the number of terms the planner will plan. The terms that survived - * were searched in full, so a hit is still a hit; a miss is not proof of - * absence. - */ - query: boolean -} - -export type SessionSearchResponse = { - hits: SessionSearchHit[] - /** - * Engine features the index on disk cannot serve, empty on a current index. - * A route ladder missing its repair rung still answers; saying so is what - * keeps the answer honest. - */ - unavailable: readonly SessionSearchUnavailableFeature[] - planner: SessionSearchPlannerReport - page: SessionSearchPage - truncated: SessionSearchTruncation - /** The index snapshot these hits came from; a cursor is only valid within it. */ - generation: number - durationMs: number -} - -export function resolveSessionSearchLimit(limit: number | undefined): number { - // Why clamped here and not at the caller: a non-positive limit becomes - // `slice(0, -1)`, which silently drops the last hit of every page. - const requested = Number.isInteger(limit) ? (limit as number) : SESSION_SEARCH_LIMIT_DEFAULT - return Math.min(Math.max(1, requested), SESSION_SEARCH_LIMIT_MAX) -} diff --git a/src/main/ai-vault-search/session-search-engine.test.ts b/src/main/ai-vault-search/session-search-engine.test.ts deleted file mode 100644 index 140f9449c3d..00000000000 --- a/src/main/ai-vault-search/session-search-engine.test.ts +++ /dev/null @@ -1,471 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import { SESSION_SEARCH_QUERY_MAX_LENGTH } from './session-search-engine-types' -import type { SessionSearchRequest, SessionSearchResponse } from './session-search-engine-types' -import { planSessionSearchQuery } from './session-search-query-planner' -import { ensureSessionSearchQuerySchema } from './session-search-query-schema' -import { EMPTY_SNIPPET, sessionSearchSnippet } from './session-search-snippet' -import { - addSyntheticSession, - markFork, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' - -let harness: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - harness = null -}) - -async function open(name: string, options = {}): Promise { - harness = await openSessionSearchHarness(name, options) - return harness -} - -function ids(result: SessionSearchResponse): string[] { - return result.hits.map((hit) => hit.sessionId) -} - -describe('the route ladder tries phrase, then AND, then repair, then OR', () => { - async function routeFor( - text: string, - request: SessionSearchRequest - ): Promise { - const { db, engine } = await open('ss-engine-route') - addSyntheticSession(db, { id: 1, text }) - return engine.search(request) - } - - it('takes the phrase route when the tokens are adjacent and in order', async () => { - const result = await routeFor('the alpha beta gamma line', { query: '"alpha beta"' }) - expect(result.planner.route).toBe('phrase') - expect(ids(result)).toEqual(['1']) - }) - - it('falls to AND when the tokens are present but not adjacent', async () => { - const result = await routeFor('beta separated alpha', { query: '"alpha beta"' }) - expect(result.planner.route).toBe('and') - expect(ids(result)).toEqual(['1']) - }) - - it('falls to OR for prose, where no phrase was ever claimed', async () => { - const result = await routeFor('the relay dropped a frame', { query: 'relay frames dropped' }) - expect(result.planner.route).toBe('or') - expect(ids(result)).toEqual(['1']) - }) - - it('repairs a typo before the OR fallback, and says which terms it changed', async () => { - const { db, engine } = await open('ss-engine-typo') - // Two copies: the repair only suggests a term the index really holds. - addSyntheticSession(db, { id: 1, text: 'the coalesces path is slow' }) - addSyntheticSession(db, { id: 2, text: 'coalesces again here' }) - const result = engine.search({ query: 'coalescs' }) - expect(result.planner.route).toBe('typo+or') - expect(result.planner.repairedTerms).toEqual(['coalesces']) - expect(ids(result).sort()).toEqual(['1', '2']) - }) - - it('keeps every term a repaired literal was typed with', async () => { - const { db, engine } = await open('ss-engine-typo-literal') - addSyntheticSession(db, { id: 1, text: 'parseJson the data' }) - addSyntheticSession(db, { id: 2, text: 'parseJson the data again' }) - // `parseJsonn(the, data)` is literal because of its punctuation; the - // corrected spelling read on its own is prose. Re-planning without carrying - // the original decision across would drop `the` and report a body that was - // never typed. - // A corrected term comes back in the index's own spelling, which unicode61 - // has folded; the terms the repair left alone keep the case they were typed. - const result = engine.search({ query: 'parseJsonn(the, data)' }) - expect(result.planner.repairedTerms).toEqual(['parsejson', 'the', 'data']) - }) - - it('does not repair a term the index already holds', async () => { - const { db, engine } = await open('ss-engine-no-typo') - addSyntheticSession(db, { id: 1, text: 'coalesces' }) - const result = engine.search({ query: 'coalesces' }) - expect(result.planner.repairedTerms).toBeUndefined() - expect(result.planner.route).toBe('or') - }) - - it('reports the scope it searched as the planner tier', async () => { - const { db, engine } = await open('ss-engine-tier') - addSyntheticSession(db, { id: 1, text: 'needle' }) - expect(engine.search({ query: 'needle' }).planner.tier).toBe('all') - expect(engine.search({ query: 'needle', scope: 'conversation' }).planner.tier).toBe( - 'conversation' - ) - }) -}) - -describe('scope picks the corpus and never switches it', () => { - async function corpus(): Promise { - const opened = await open('ss-engine-scope') - addSyntheticSession(opened.db, { id: 1, text: 'harbor pilot manifest', role: 'user' }) - addSyntheticSession(opened.db, { id: 2, text: 'harbor tool output line', role: 'tool' }) - return opened - } - - it('searches conversation turns only under `conversation`', async () => { - const { engine } = await corpus() - expect(ids(engine.search({ query: 'harbor', scope: 'conversation' }))).toEqual(['1']) - }) - - it('includes tool output under `all`, which is the default', async () => { - const { engine } = await corpus() - expect(ids(engine.search({ query: 'harbor', scope: 'all' })).sort()).toEqual(['1', '2']) - expect(ids(engine.search({ query: 'harbor' })).sort()).toEqual(['1', '2']) - }) - - it('returns nothing rather than widening when the narrow scope misses', async () => { - // The panel's two-tier typing is a UI policy (PR 7). An engine that widened - // here would make a result impossible to reproduce from its own request. - const { engine } = await corpus() - const result = engine.search({ query: 'output', scope: 'conversation' }) - expect(result.hits).toEqual([]) - expect(result.planner.tier).toBe('conversation') - }) - - it('matches an identifier through its pieces only in the full corpus', async () => { - const { db, engine } = await open('ss-engine-identifiers') - addSyntheticSession(db, { id: 1, text: 'resolveTerminalPath' }) - // The identifier shadow column lives in messages_fts alone. - expect(ids(engine.search({ query: 'terminal path' }))).toEqual(['1']) - expect(engine.search({ query: 'terminal path', scope: 'conversation' }).hits).toEqual([]) - }) -}) - -describe('the conversation scope is a column filter, and it binds the whole query', () => { - it('refuses an AND whose second term lives only in tool output', async () => { - // The filter binds to the expression it prefixes. `{cols}: (a AND b)` - // filters both terms; `{cols}: a AND b` filters only `a` and searches tool - // output for the rest, which is a conversation search answering from a - // column it promised not to read. - const { db, engine } = await open('ss-engine-scope-binding') - addSyntheticSession(db, { id: 1, text: 'alpha gamma beta' }) - addSyntheticSession(db, { id: 2, text: 'alpha gamma', toolText: 'beta' }) - // Quoted, so the query is literal; not adjacent, so the phrase rung misses - // and the AND rung is the one that answers. - const query = '"alpha" beta' - - const wide = engine.search({ query, scope: 'all' }) - expect(wide.planner.route).toBe('and') - expect(ids(wide).sort()).toEqual(['1', '2']) - - const narrowed = engine.search({ query, scope: 'conversation' }) - expect(narrowed.planner.route).toBe('and') - expect(ids(narrowed)).toEqual(['1']) - }) - - it('ranks a conversation hit down for tool output it will not show', async () => { - // The one behavioural difference the column filter carries, pinned rather - // than wished away. FTS5's bm25 normalises by the whole row's length and - // has no per-column length, so two rows with identical prose do not score - // identically when one of them also holds tool output. A dedicated - // two-column table scored them the same. The rowid set is unchanged, which - // is what the decision was measured on; the order within it can move. - const { db, engine } = await open('ss-engine-scope-weights') - addSyntheticSession(db, { id: 1, text: 'harbor pilot' }) - addSyntheticSession(db, { id: 2, text: 'harbor pilot', toolText: 'unrelated '.repeat(40) }) - const narrowed = engine.search({ query: 'harbor', scope: 'conversation' }) - expect(ids(narrowed)).toEqual(['1', '2']) - expect(narrowed.hits[0]!.score).toBeGreaterThan(narrowed.hits[1]!.score) - }) - - it('never snippets a conversation hit out of tool output', async () => { - const { db, engine } = await open('ss-engine-scope-snippet') - addSyntheticSession(db, { id: 1, text: 'harbor pilot', toolText: 'harbor tool output line' }) - const [hit] = engine.search({ query: 'harbor', scope: 'conversation' }).hits - expect(hit?.evidence?.snippet).toContain('pilot') - expect(hit?.evidence?.snippet).not.toContain('output') - // And asked for a tool-only row directly, it has nothing to show. - addSyntheticSession(db, { id: 2, text: 'harbor tool output line', role: 'tool' }) - const rowid = Number( - (db.prepare('SELECT max(id) AS id FROM messages').get() as { id: number }).id - ) - const plan = planSessionSearchQuery('harbor') - expect(sessionSearchSnippet(db, 'conversation', rowid, plan)).toEqual(EMPTY_SNIPPET) - expect(sessionSearchSnippet(db, 'all', rowid, plan).text).toContain('output') - }) -}) - -describe('a session is one hit, however many of its rows matched', () => { - it.each(['relevance', 'newest'] as const)( - 'keeps a short session on the %s page beside a 650-row session', - async (sort) => { - const { db, engine } = await open('ss-engine-aggregate', { sessionCandidateLimit: 600 }) - addSyntheticSession(db, { id: 1, rows: 650, updatedAt: '2026-09-06T00:00:00.000Z' }) - addSyntheticSession(db, { - id: 2, - text: 'needle padding', - updatedAt: '2026-09-05T00:00:00.000Z' - }) - // Collapsing to one row per session happens before the candidate limit, - // so the 650-row session cannot crowd the one-row session off the page on - // either order; which of them ranks first is the sort's business. - expect(ids(engine.search({ query: 'needle', filters: { sort } })).sort()).toEqual(['1', '2']) - } - ) - - it('folds forks the same way for an operator-only page as for a text page', async () => { - const { db, engine } = await open('ss-engine-forks') - for (const id of [1, 2, 3, 4]) { - addSyntheticSession(db, { id, updatedAt: `2026-09-0${id}T00:00:00.000Z` }) - } - markFork(db, [1, 2, 3, 4], 'shared-fork-prefix') - const operatorOnly = engine.search({ query: 'repo:app' }) - const withText = engine.search({ query: 'needle repo:app' }) - expect(ids(operatorOnly)).toEqual(['4']) - expect(operatorOnly.hits[0]?.duplicateCount).toBe(4) - expect(ids(withText)).toEqual(ids(operatorOnly)) - expect(withText.hits[0]?.duplicateCount).toBe(4) - }) - - it('answers an operator-only query with the newest sessions and no evidence', async () => { - const { db, engine } = await open('ss-engine-operator-only') - addSyntheticSession(db, { id: 1, updatedAt: '2026-09-01T00:00:00.000Z' }) - addSyntheticSession(db, { id: 2, updatedAt: '2026-09-09T00:00:00.000Z' }) - const result = engine.search({ query: 'repo:app' }) - expect(ids(result)).toEqual(['2', '1']) - expect(result.hits[0]?.evidence).toBeNull() - }) - - it('has no hits for a query with neither text nor operators', async () => { - const { db, engine } = await open('ss-engine-empty') - addSyntheticSession(db, { id: 1 }) - expect(engine.search({ query: ' ' }).hits).toEqual([]) - }) -}) - -describe('filters narrow retrieval, not just the page', () => { - it('finds a scoped match behind 600 out-of-scope rows', async () => { - const { db, engine } = await open('ss-engine-scoped') - addSyntheticSession(db, { id: 1, cwd: '/unrelated', rows: 600 }) - addSyntheticSession(db, { id: 2, cwd: '/target', text: 'needle padding' }) - expect(ids(engine.search({ query: 'needle', filters: { scopePaths: ['/target'] } }))).toEqual([ - '2' - ]) - }) - - it('falls back to a later rung when the exact hit is out of scope', async () => { - const { db, engine } = await open('ss-engine-scoped-route') - addSyntheticSession(db, { id: 1, cwd: '/unrelated', text: 'resolveTerminalPath' }) - addSyntheticSession(db, { id: 2, cwd: '/target', text: 'resolve terminal path' }) - expect( - ids(engine.search({ query: 'resolveTerminalPath', filters: { scopePaths: ['/target'] } })) - ).toEqual(['2']) - }) -}) - -describe('evidence', () => { - it('takes each snippet from that hit’s own best message', async () => { - const { db, engine } = await open('ss-engine-snippet') - // Written first, so its row owns the lowest rowid: the row a dropped rowid - // constraint would hand back for every hit. - addSyntheticSession(db, { - id: 1, - text: 'hydration marmoset appears once in a long paragraph about routing and caching', - updatedAt: '2026-09-01T00:00:00.000Z' - }) - addSyntheticSession(db, { - id: 2, - text: 'hydration capybara', - updatedAt: '2026-09-09T00:00:00.000Z' - }) - const hits = engine.search({ query: 'hydration' }).hits - expect(hits[0]?.evidence?.snippet).toContain('capybara') - expect(hits[0]?.evidence?.snippet).not.toContain('marmoset') - expect(hits.find((hit) => hit.sessionId === '1')?.evidence?.snippet).toContain('marmoset') - }) - - it('shows the prose column rather than the identifier shadow when both match', async () => { - const { db, engine } = await open('ss-engine-snippet-shadow') - addSyntheticSession(db, { - id: 1, - text: 'resolveTerminalPath is broken and the terminal never comes up for a pane, which is odd because every other pane on this host resolves its path' - }) - const snippet = engine.search({ query: 'terminal path' }).hits[0]?.evidence?.snippet ?? '' - expect(snippet).toContain('[[') - expect(snippet).not.toContain('resolve [[terminal]] [[path]]') - }) - - it('flags a snippet it had to cut, and counts it on the result', async () => { - const { db, engine } = await open('ss-engine-snippet-truncated') - // The window is twelve tokens wide, and one of them is 4000 characters, so - // the token count is no bound at all on what a hit carries. - addSyntheticSession(db, { id: 1, text: `needle ${'x'.repeat(4000)}` }) - const result = engine.search({ query: 'needle' }) - expect(result.hits[0]?.evidence?.snippetTruncated).toBe(true) - expect(result.hits[0]?.evidence?.snippet.length).toBeLessThan(600) - expect(result.truncated.snippets).toBe(1) - }) - - it('leaves an ordinary snippet unflagged', async () => { - const { db, engine } = await open('ss-engine-snippet-whole') - addSyntheticSession(db, { id: 1, text: 'needle in a short line' }) - const result = engine.search({ query: 'needle' }) - expect(result.hits[0]?.evidence?.snippetTruncated).toBeUndefined() - expect(result.truncated.snippets).toBe(0) - }) -}) - -describe('source presence comes from the files table, never a stat', () => { - it('calls a session with a live file record present', async () => { - const { db, engine } = await open('ss-engine-presence') - addSyntheticSession(db, { id: 1 }) - expect(engine.search({ query: 'needle' }).hits[0]?.source).toBe('present') - }) - - it('calls a session with no file record unverifiable, and still returns it', async () => { - // Loss of contact is never evidence of absence: the hit stays on the page. - const { db, engine } = await open('ss-engine-presence-unknown') - addSyntheticSession(db, { id: 1, filePath: null }) - const hits = engine.search({ query: 'needle' }).hits - expect(hits).toHaveLength(1) - expect(hits[0]?.source).toBe('unverifiable') - }) -}) - -describe('the engine carries its own schema and puts it back', () => { - it('installs the vocabulary and the log over an index a writer built alone', async () => { - // The store creates none of these: PR 3's indexer can fill a whole index - // before anything opens an engine over it. - const { db, engine } = await open('ss-engine-installs') - addSyntheticSession(db, { id: 1, text: 'the coalesces path is slow' }) - addSyntheticSession(db, { id: 2, text: 'coalesces again here' }) - const result = engine.search({ query: 'coalescs' }) - expect(result.unavailable).toEqual([]) - expect(result.planner.route).toBe('typo+or') - expect(ids(result).sort()).toEqual(['1', '2']) - }) - - it('re-creates a vocabulary that vanished under a live engine', async () => { - const { db, engine } = await open('ss-engine-vocab-vanishes') - addSyntheticSession(db, { id: 1, text: 'coalesces here now' }) - addSyntheticSession(db, { id: 2, text: 'coalesces again here' }) - expect(engine.search({ query: 'coalescs' }).planner.route).toBe('typo+or') - - db.exec('DROP TABLE messages_vocab') - const after = engine.search({ query: 'coalescs' }) - expect(after.unavailable).toEqual([]) - expect(after.planner.route).toBe('typo+or') - }) - - it('names the feature it cannot serve when the vocabulary has no source left', async () => { - // What an index being rebuilt by another handle looks like from here. The - // vocabulary can be created over a missing `messages_fts` and every query - // against it then fails, so the probe reads the source, not the view. - // - // With one FTS table there is no scope left to answer from, so this is now - // the boundary of the degrade: the engine names the feature and the search - // fails loudly on the table it cannot read, rather than returning an empty - // page that looks like an answer. - const { db, engine } = await open('ss-engine-vocab-source-gone') - addSyntheticSession(db, { id: 1, text: 'coalesces here now', role: 'user' }) - db.exec('DROP TABLE messages_vocab; DROP TABLE messages_fts') - - expect(ensureSessionSearchQuerySchema(db)).toEqual(['typo-repair']) - for (const scope of ['all', 'conversation'] as const) { - expect(() => engine.search({ query: 'coalesces', scope })).toThrow(/no such (fts5 )?table/i) - } - }) - - it('picks the feature back up when the source comes back', async () => { - const { db, engine } = await open('ss-engine-vocab-returns') - addSyntheticSession(db, { id: 1, text: 'coalesces here now' }) - addSyntheticSession(db, { id: 2, text: 'coalesces again here' }) - const fts = ( - db.prepare("SELECT sql FROM sqlite_master WHERE name = 'messages_fts'").get() as { - sql: string - } - ).sql - db.exec('DROP TABLE messages_vocab; DROP TABLE messages_fts') - expect(ensureSessionSearchQuerySchema(db)).toEqual(['typo-repair']) - - db.exec(fts) - // Two, because the vocabulary only offers a term at least two rows carry. - addSyntheticSession(db, { id: 3, text: 'coalesces one more time' }) - addSyntheticSession(db, { id: 4, text: 'coalesces once again' }) - // Nothing throws on the way back up, so the recovery cannot come from the - // error path; it comes from the probe running per search. - const restored = engine.search({ query: 'coalescs' }) - expect(restored.unavailable).toEqual([]) - expect(restored.planner.route).toBe('typo+or') - }) -}) - -describe('a query the engine had to cut says so', () => { - it('answers a query whose cap falls inside an astral character', async () => { - // The cut is on a whole code point rather than a code unit, so nothing - // downstream is handed half a surrogate pair. That is hygiene rather than a - // behaviour: the planner's tokenizer does not treat a lone surrogate as a - // token character, so it drops out of the terms either way. What this pins - // is that the boundary is answerable at all. - const { db, engine } = await open('ss-engine-surrogate-cap') - const kept = 'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH - 2) - addSyntheticSession(db, { id: 1, text: kept }) - const result = engine.search({ query: `${kept} 😀 tail` }) - expect(result.truncated.query).toBe(true) - expect(result.hits.map((hit) => hit.sessionId)).toEqual(['1']) - }) - - it('loads a candidate set larger than one batch of bound ids', async () => { - // The id list is as long as the candidate limit and every id is a bound - // parameter. No SQLite this stack can run refuses 1,100 of them, so this - // pins that batching returns the same answer, not that it rescues one. - const { db, engine } = await open('ss-engine-id-batching', { - sessionCandidateLimit: 1200 - }) - for (let id = 1; id <= 1100; id++) { - addSyntheticSession(db, { id, text: 'needle' }) - } - const result = engine.search({ query: 'needle', limit: 5 }) - expect(result.hits).toHaveLength(5) - expect(result.truncated.candidates).toBe(false) - }) - - it('reports truncation when the planner drops terms past its cap', async () => { - // The 56th term is the only one that matches. Without the flag this is a - // confident empty answer to a query the engine never finished reading. - const { db, engine } = await open('ss-engine-term-cap') - addSyntheticSession(db, { id: 1, text: 'onlyattheend' }) - const query = `${Array.from({ length: 55 }, (_unused, n) => `term${n}`).join(' ')} onlyattheend` - const result = engine.search({ query }) - expect(result.hits).toEqual([]) - expect(result.truncated.query).toBe(true) - }) - - it('reports truncation when the query is longer than the engine will plan', async () => { - const { db, engine } = await open('ss-engine-length-cap') - addSyntheticSession(db, { id: 1, text: 'needle' }) - const result = engine.search({ query: `needle ${'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH)}` }) - expect(result.truncated.query).toBe(true) - }) - - it('claims no truncation for a query that fit', async () => { - const { db, engine } = await open('ss-engine-no-cap') - addSyntheticSession(db, { id: 1, text: 'needle' }) - expect(engine.search({ query: 'needle' }).truncated.query).toBe(false) - }) -}) - -describe('a query longer than the engine will plan is cut, not refused', () => { - it('cuts one enormous token down to the cap before FTS5 ever sees it', async () => { - const { db, engine } = await open('ss-engine-long-query') - // The planner already caps how many terms it will plan, so a long query of - // ordinary words is bounded without this. What is not bounded is a single - // token: one 100 kB word is one term, and FTS5 would carry the whole thing - // into the MATCH expression. The cut is observable because the indexed - // token is exactly the capped length. - addSyntheticSession(db, { id: 1, text: 'x'.repeat(SESSION_SEARCH_QUERY_MAX_LENGTH) }) - expect(ids(engine.search({ query: 'x'.repeat(4000) }))).toEqual(['1']) - }) -}) - -describe('unicode terms survive the round trip', () => { - it.each(['café', 'C', 'R', 'x', '修復', '안녕하세요'])('searches %s', async (text) => { - const { db, engine } = await open('ss-engine-unicode') - addSyntheticSession(db, { id: 1, text }) - expect(engine.search({ query: text }).hits).toHaveLength(1) - }) -}) diff --git a/src/main/ai-vault-search/session-search-engine.ts b/src/main/ai-vault-search/session-search-engine.ts deleted file mode 100644 index 4b6e11e407d..00000000000 --- a/src/main/ai-vault-search/session-search-engine.ts +++ /dev/null @@ -1,349 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' -import type { TranscriptMessageRole } from '../ai-vault/session-transcript-consumers' -import { sliceAtCodeUnitLimit } from '../ai-vault/session-scanner-text-normalization' -import { - hasAiVaultSearchQueryOperators, - splitAiVaultSearchQuery, - type AiVaultSearchQuerySplit -} from '../../shared/ai-vault-search-query-operators' -import { matchesAiVaultQueryOperators } from '../../shared/ai-vault-session-filters' -import { - resolveSessionSearchLimit, - SESSION_SEARCH_QUERY_MAX_LENGTH, - type SessionSearchHit, - type SessionSearchRequest, - type SessionSearchResponse, - type SessionSearchScope, - type SessionSearchSourcePresence -} from './session-search-engine-types' -import { readIndexGeneration } from './session-search-index-generation' -import { - rankSessionHits, - type MessageRow, - type RankedSession, - type SessionRow -} from './session-search-hit-ranking' -import { - decodeSessionSearchCursor, - encodeSessionSearchCursor, - sessionSearchPageKey -} from './session-search-page-cursor' -import { planSessionSearchQuery } from './session-search-query-planner' -import { logSessionSearchQuery } from './session-search-query-log' -import { - SessionSearchRetrieval, - type RetrievalScope, - type Retrieved -} from './session-search-retrieval' -import { sessionRowFilter } from './session-search-row-filter' -import { - ensureSessionSearchQuerySchema, - type SessionSearchUnavailableFeature -} from './session-search-query-schema' -import { EMPTY_SNIPPET, sessionSearchSnippet } from './session-search-snippet' -import { sessionSourcePresence } from './session-search-source-presence' - -/** - * Sessions retrieved before ranking cuts the page. - * - * Not a fixed constant (the reviewer's F13): it is the knob that trades page - * completeness for retrieval cost, and the right value depends on index size. - * Measurements behind this default, and what changing it costs, are in - * docs/reference/agent-session-search-query-tuning.md. - */ -export const SESSION_SEARCH_CANDIDATE_LIMIT_DEFAULT = 600 - -/** One ranked list plus what produced it; a page is a slice of `ranked`. */ -type RankedPage = { - ranked: RankedSession[] - /** Null when no text was searched, so there is nothing to snippet from. */ - retrieved: Retrieved | null - /** - * Retrieval may have missed a session: a cap ended it, not the data. True - * whether the candidate limit filled or the operator walk gave up scanning. - */ - incomplete: boolean -} - -export type SessionSearchEngineOptions = { - sessionCandidateLimit?: number - /** Oldest transcript mtime a hit may come from; PR 3 derives it from retention. */ - retentionCutoffMs?: number | null - /** Write each query to `search_log`. Off unless a caller asks (see query-log). */ - logQueries?: boolean -} - -/** - * Ranked session search over the PR 2 index. - * - * A library: it holds no timers, reads no settings, and knows nothing about - * Electron, IPC or a panel. It is handed a connection rather than opening one, - * because which process may open, rebuild or unlink the index file is PR 3b's - * decision and not a query engine's. - * - * **Every read here is a single statement, and no read transaction is ever - * open across an `await`.** There is no `BEGIN` on this path, no `.iterate()` - * outliving its statement, and `search` is synchronous end to end. That is a - * constraint PR 2 measured rather than a style: a reader that pins a WAL - * snapshot holds off every checkpoint behind it, and the same 47 MB of writes - * that leave a 9.9 MB WAL grew to 266 MB with one `BEGIN` + `SELECT` held open. - * - * One search is one synchronous pass, and every page of it is a slice of the - * same ranked list. That list is rebuilt per page rather than streamed, which - * is what makes a page repeatable: within one index generation the same request - * ranks the same way, and a cursor from any other generation is refused. - * - * That fence is strict on purpose, and the cost is worth stating plainly: any - * committed read moves the generation, so while a backfill is running an - * outstanding cursor will be refused, often within a second. Pagination is - * usable against a settled index and unreliable against one still filling. The - * rejection carries both generations, so a caller that sees `stale-generation` - * knows the index moved rather than that it holds a bad cursor, and can quietly - * re-issue page one instead of showing anyone an error. - */ -export class SessionSearchEngine { - private retrieval: SessionSearchRetrieval - private readonly candidateLimit: number - /** Re-probed whenever a query proves it stale; see `withCapabilityRetry`. */ - private unavailable: readonly SessionSearchUnavailableFeature[] - - constructor( - private readonly db: SyncDatabase, - private readonly options: SessionSearchEngineOptions = {} - ) { - this.candidateLimit = options.sessionCandidateLimit ?? SESSION_SEARCH_CANDIDATE_LIMIT_DEFAULT - // Installed here and not on the first search, so the generation triggers are - // watching before anything this engine will be asked to page over is - // written, and so retrieval below prepares against tables that exist. - this.unavailable = ensureSessionSearchQuerySchema(this.db) - this.retrieval = new SessionSearchRetrieval(this.db, !this.unavailable.includes('typo-repair')) - } - - search(request: SessionSearchRequest): SessionSearchResponse { - const startedAt = performance.now() - this.probeCapabilities() - const generation = readIndexGeneration(this.db) - const scope = request.scope ?? 'all' - const sort = request.filters?.sort ?? 'relevance' - // Not a bare `slice`: cutting between a surrogate pair leaves a lone half - // that no tokenizer can match and that a caller cannot echo back. - const capped = sliceAtCodeUnitLimit(request.query, SESSION_SEARCH_QUERY_MAX_LENGTH) - const split = splitAiVaultSearchQuery(capped) - const retrievalScope: RetrievalScope = { - scope, - sort, - filter: sessionRowFilter(request.filters ?? {}, this.options.retentionCutoffMs ?? null), - matchesOperators: operatorPredicate(split), - candidateLimit: this.candidateLimit - } - // Decoded before any retrieval: a cursor the engine will refuse must not - // cost a query, and the caller has to hear about it either way. - const pageKey = sessionSearchPageKey(request) - const offset = request.cursor - ? decodeSessionSearchCursor(request.cursor, generation, pageKey) - : 0 - - const plan = planSessionSearchQuery(split.text) - const { ranked, retrieved, incomplete } = this.withCapabilityRetry(() => - plan.terms.length === 0 - ? this.operatorOnly(split, retrievalScope) - : this.text(plan, retrievalScope, sort) - ) - - const limit = resolveSessionSearchLimit(request.limit) - const page = ranked.slice(offset, offset + limit) - const hits = this.hits(page, scope, retrieved) - const hasMore = ranked.length > offset + limit - const response: SessionSearchResponse = { - hits, - unavailable: this.unavailable, - planner: { - route: retrieved?.route ?? 'or', - tier: scope, - ...(retrieved?.repairedTerms ? { repairedTerms: retrieved.repairedTerms } : {}) - }, - page: { - hasMore, - cursor: hasMore ? encodeSessionSearchCursor(generation, offset + limit, pageKey) : null - }, - truncated: { - // Decided by retrieval, which is the only layer that knows whether a cap - // ended it. Deriving it from the hits cannot work: an operator walk that - // gave up at its scan ceiling returns no hits, and so does a search that - // genuinely matched nothing. - candidates: incomplete, - snippets: hits.filter((hit) => hit.evidence?.snippetTruncated).length, - query: capped.length < request.query.length || plan.truncated - }, - generation, - durationMs: performance.now() - startedAt - } - if (this.options.logQueries) { - logSessionSearchQuery(this.db, { - query: request.query, - route: response.planner.route, - hits: hits.length, - durationMs: response.durationMs - }) - } - return response - } - - /** - * Where the engine's own schema is created and checked, once per search. - * - * A capability is a fact about the file, not about this object: another handle - * can rebuild the index under a live connection, so a verdict cached in the - * constructor is wrong for the rest of the engine's life in both directions — - * it would keep reaching for a table that went away, and never pick one back - * up when it returned. Retrieval is only rebuilt when the answer changes, so - * the steady-state cost is one indexed lookup and nothing else. - */ - private probeCapabilities(): void { - const unavailable = ensureSessionSearchQuerySchema(this.db) - if (unavailable.join() === this.unavailable.join()) { - return - } - this.unavailable = unavailable - this.retrieval = new SessionSearchRetrieval(this.db, !unavailable.includes('typo-repair')) - } - - /** - * Runs a retrieval, and re-probes once if it turns out the index no longer - * has what an earlier probe found. - * - * `probeCapabilities` already runs per search, so this only covers the window - * between that probe and the statement that reaches for the table. Losing a - * table there is a thrown error rather than a wrong verdict, so it re-probes - * and runs the search again. - */ - private withCapabilityRetry(run: () => RankedPage): RankedPage { - try { - return run() - } catch (error) { - if (!isMissingTableError(error)) { - throw error - } - this.probeCapabilities() - return run() - } - } - - /** - * Operators with no free text still name a scope, so the answer is the newest - * sessions inside it. Ranked through the same path as a text query, because - * forks must fold here exactly as they do there or the same sessions answer - * `repo:x` and `word repo:x` differently. There is no relevance signal - * without text, so the order is always newest. - */ - private operatorOnly(split: AiVaultSearchQuerySplit, scope: RetrievalScope): RankedPage { - if (!hasAiVaultSearchQueryOperators(split)) { - return { ranked: [], retrieved: null, incomplete: false } - } - const { sessions, incomplete } = this.retrieval.recent(scope) - return { ranked: rankSessionHits(sessions, new Map(), 'newest'), retrieved: null, incomplete } - } - - private text( - plan: ReturnType, - scope: RetrievalScope, - sort: 'relevance' | 'newest' - ): RankedPage { - const retrieved = this.retrieval.run(plan, scope) - // `match` already grouped to one best row per session. - const best = new Map(retrieved.rows.map((row) => [row.session_row_id, row])) - // Operators cut here, after retrieval, so the candidate count still reports - // what the SQL limit saw: that is what tells a caller the limit was binding. - const sessions = this.retrieval.loadSessions([...best.keys()], scope) - // Counted before the operator predicate and before fork folding: the SQL - // LIMIT is what could have hidden a session, and it saw the unfiltered set. - return { - ranked: rankSessionHits(sessions, best, sort), - retrieved, - incomplete: best.size >= this.candidateLimit - } - } - - /** Snippets and source presence are paid for by the page, never by the list. */ - private hits( - page: readonly RankedSession[], - scope: SessionSearchScope, - retrieved: Retrieved | null - ): SessionSearchHit[] { - const presence = sessionSourcePresence( - this.db, - page.map((entry) => entry.session.id) - ) - return page.map((entry) => this.hit(entry, scope, retrieved, presence)) - } - - private hit( - entry: RankedSession, - scope: SessionSearchScope, - retrieved: Retrieved | null, - presence: ReadonlyMap - ): SessionSearchHit { - const { session, message } = entry - const snippet = - message && retrieved - ? sessionSearchSnippet(this.db, scope, message.rowid, retrieved.plan) - : EMPTY_SNIPPET - return { - ...sessionFields(session), - score: entry.score, - ...(entry.duplicateCount > 1 ? { duplicateCount: entry.duplicateCount } : {}), - source: presence.get(session.id) ?? 'unverifiable', - evidence: message - ? { - role: message.role as TranscriptMessageRole, - timestamp: message.ts, - snippet: snippet.text, - ...(snippet.truncated ? { snippetTruncated: true } : {}) - } - : null - } - } -} - -// SQLite reports a table that went away at the statement that reaches for it. -// `fts5` is in the message when the table is the vocabulary's target, which is -// the one an index rebuilt under a live connection loses first. -const MISSING_TABLE = /no such (fts5 )?table/i - -function isMissingTableError(error: unknown): boolean { - return error instanceof Error && MISSING_TABLE.test(error.message) -} - -/** - * The one reading of `repo:` / `path:`: the sessions panel's own predicate, over - * the columns the index stores. The engine has no project map, so a session's - * repo label falls back to its folder label, which is what the panel does for - * every session it cannot resolve a project for. - */ -function operatorPredicate(split: AiVaultSearchQuerySplit): (session: SessionRow) => boolean { - if (!hasAiVaultSearchQueryOperators(split)) { - return () => true - } - return (session) => - matchesAiVaultQueryOperators( - { cwd: session.cwd, filePath: session.file_path }, - { repoTerms: split.repoTerms, pathTerms: split.pathTerms } - ) -} - -function sessionFields( - session: SessionRow -): Omit { - return { - agent: session.agent, - sessionId: session.session_id, - filePath: session.file_path, - codexHome: session.codex_home, - title: session.title, - cwd: session.cwd, - branch: session.branch, - updatedAt: session.updated_at, - messageCount: session.message_count, - resumeCommand: session.resume_command - } -} diff --git a/src/main/ai-vault-search/session-search-fts5-contract.test.ts b/src/main/ai-vault-search/session-search-fts5-contract.test.ts deleted file mode 100644 index be815623ce7..00000000000 --- a/src/main/ai-vault-search/session-search-fts5-contract.test.ts +++ /dev/null @@ -1,172 +0,0 @@ -import { mkdtemp } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, describe, expect, it } from 'vitest' -import { removeTree } from '../../shared/windows-transient-lock-removal' -import type SyncDatabase from '../sqlite/sync-database' -import { indexTokens } from './session-search-query-planner' -import { ensureSessionSearchQuerySchema } from './session-search-query-schema' -import { openSessionSearchDatabase } from './session-search-schema' - -// SQLite/FTS5 behaviours the query layer depends on. Each one cost a live -// debugging session; a refactor that reintroduces the trap fails here. - -const FIRST_ROWID = 101 -const SECOND_ROWID = 202 - -let tempRoots: string[] = [] - -afterEach(async () => { - await Promise.all(tempRoots.map((root) => removeTree(root))) - tempRoots = [] -}) - -async function openDatabase(): Promise { - const root = await mkdtemp(join(tmpdir(), 'orca-fts5-contract-')) - tempRoots.push(root) - return openSessionSearchDatabase(join(root, 'index.sqlite')) -} - -function insertMessageRow(db: SyncDatabase, rowid: number, text: string): void { - db.prepare( - `INSERT INTO messages_fts(rowid, user_text, assistant_text, tool_text, identifiers) - VALUES (?, ?, '', '', '')` - ).run(rowid, text) -} - -describe('FTS5 aux functions take the table name, never an alias', () => { - it('rejects bm25 over an aliased table and accepts the table-name form', async () => { - const db = await openDatabase() - insertMessageRow(db, FIRST_ROWID, 'alpha marmoset one') - - expect(() => - db.prepare('SELECT bm25(f) AS score FROM messages_fts f WHERE f MATCH ?').all('alpha') - ).toThrow(/no such column: f/) - - const scored = db - .prepare('SELECT bm25(messages_fts) AS score FROM messages_fts WHERE messages_fts MATCH ?') - .all('alpha') as { score: number }[] - expect(scored).toHaveLength(1) - expect(Number.isFinite(scored[0]?.score)).toBe(true) - db.close() - }) - - it('rejects snippet over an aliased table too', async () => { - const db = await openDatabase() - insertMessageRow(db, FIRST_ROWID, 'alpha marmoset one') - - expect(() => - db - .prepare( - "SELECT snippet(f, -1, '[', ']', '…', 12) AS s FROM messages_fts f WHERE f MATCH ?" - ) - .all('alpha') - ).toThrow(/no such column: f/) - db.close() - }) -}) - -describe('a rowid constraint beside MATCH is honoured only as a subselect', () => { - it('ignores `rowid = ?` and returns every match, first row first', async () => { - const db = await openDatabase() - insertMessageRow(db, FIRST_ROWID, 'alpha marmoset one') - insertMessageRow(db, SECOND_ROWID, 'alpha capybara two') - - const rows = db - .prepare('SELECT rowid FROM messages_fts WHERE messages_fts MATCH ? AND rowid = ?') - .all('alpha', SECOND_ROWID) as { rowid: number }[] - // The planner drops the constraint entirely: both rows come back. - expect(rows.map((row) => row.rowid)).toEqual([FIRST_ROWID, SECOND_ROWID]) - // A caller reading one row therefore gets the first match, not the one asked for. - const single = db - .prepare('SELECT rowid FROM messages_fts WHERE messages_fts MATCH ? AND rowid = ?') - .get('alpha', SECOND_ROWID) as { rowid: number } | undefined - expect(single?.rowid).toBe(FIRST_ROWID) - db.close() - }) - - it('ignores `rowid IN (?)` the same way', async () => { - const db = await openDatabase() - insertMessageRow(db, FIRST_ROWID, 'alpha marmoset one') - insertMessageRow(db, SECOND_ROWID, 'alpha capybara two') - - const rows = db - .prepare('SELECT rowid FROM messages_fts WHERE messages_fts MATCH ? AND rowid IN (?)') - .all('alpha', SECOND_ROWID) as { rowid: number }[] - expect(rows.map((row) => row.rowid)).toEqual([FIRST_ROWID, SECOND_ROWID]) - db.close() - }) - - it('honours `rowid IN (SELECT ?)` even with the session join on', async () => { - const db = await openDatabase() - db.prepare( - `INSERT INTO sessions(id,agent,session_id,file_path,title,resume_command) - VALUES (1,'claude','1','/synthetic/1','fixture','')` - ).run() - for (const rowid of [FIRST_ROWID, SECOND_ROWID]) { - db.prepare("INSERT INTO messages(id,session_row_id,role) VALUES (?,1,'user')").run(rowid) - } - insertMessageRow(db, FIRST_ROWID, 'alpha marmoset one') - insertMessageRow(db, SECOND_ROWID, 'alpha capybara two') - - // The shape the snippet read uses: the joins are what subtract a row whose - // session a purge cut loose, and they must not cost the rowid constraint - // its effect. - const snippet = db - .prepare( - `SELECT snippet(messages_fts, -1, '[', ']', '…', 12) AS s - FROM messages_fts - JOIN messages m ON m.id = messages_fts.rowid - JOIN sessions s ON s.id = m.session_row_id - WHERE messages_fts MATCH ? AND messages_fts.rowid IN (SELECT ?)` - ) - .get('alpha', SECOND_ROWID) as { s: string } | undefined - expect(snippet?.s).toContain('capybara') - expect(snippet?.s).not.toContain('marmoset') - db.close() - }) -}) - -describe('sessions.file_path is deliberately not unique', () => { - it('accepts two sessions sharing one store path', async () => { - const db = await openDatabase() - const insert = db.prepare( - `INSERT INTO sessions(agent, session_id, file_path, title, resume_command) - VALUES (?, ?, ?, ?, ?)` - ) - // OpenCode and Cursor keep every session in one SQLite store; files.path is the key. - const storePath = '/home/user/.local/share/opencode/storage.db' - insert.run('opencode', 'ses_one', storePath, 'first', 'opencode --session ses_one') - expect(() => - insert.run('opencode', 'ses_two', storePath, 'second', 'opencode --session ses_two') - ).not.toThrow() - - const rows = db - .prepare('SELECT session_id FROM sessions WHERE file_path = ? ORDER BY session_id') - .all(storePath) as { session_id: string }[] - expect(rows.map((row) => row.session_id)).toEqual(['ses_one', 'ses_two']) - db.close() - }) -}) - -describe('the planner tokenizer draws the same boundaries as unicode61', () => { - // unicode61 folds case and strips Latin diacritics on both index and query side. - function asIndexed(token: string): string { - return token.toLowerCase().normalize('NFD').replaceAll(/\p{M}/gu, '') - } - - it('produces exactly the terms fts5vocab reports for the same text', async () => { - const db = await openDatabase() - // The vocabulary is the engine's own object, not the store's. - ensureSessionSearchQuerySchema(db) - const corpus = - 'resolveTerminalPath src/main/foo-bar.ts a.b C++ #123 修复 café naïve MAX_TOKEN x' - insertMessageRow(db, FIRST_ROWID, corpus) - const indexed = ( - db.prepare('SELECT term FROM messages_vocab ORDER BY term').all() as { term: string }[] - ).map((row) => row.term) - - expect([...new Set(indexTokens(corpus).map(asIndexed))].sort()).toEqual(indexed) - db.close() - }) -}) diff --git a/src/main/ai-vault-search/session-search-hit-ranking.test.ts b/src/main/ai-vault-search/session-search-hit-ranking.test.ts deleted file mode 100644 index 54919bace0f..00000000000 --- a/src/main/ai-vault-search/session-search-hit-ranking.test.ts +++ /dev/null @@ -1,102 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { rankSessionHits, type MessageRow, type SessionRow } from './session-search-hit-ranking' - -function session(id: number, overrides: Partial = {}): SessionRow { - return { - id, - agent: 'claude', - session_id: String(id), - file_path: `/synthetic/${id}.jsonl`, - codex_home: null, - title: 'fixture', - cwd: '/repo/app', - branch: null, - updated_at: '2026-09-01T00:00:00.000Z', - message_count: 1, - resume_command: 'resume', - content_hash: null, - content_hash_count: 0, - ...overrides - } -} - -function match(id: number, score: number): MessageRow { - return { rowid: id, score, session_row_id: id, role: 'user', ts: null } -} - -function matches(...rows: MessageRow[]): Map { - return new Map(rows.map((row) => [row.session_row_id, row])) -} - -describe('order', () => { - it('ranks by score under relevance and by recency under newest', () => { - const sessions = [ - session(1, { updated_at: '2026-09-01T00:00:00.000Z' }), - session(2, { updated_at: '2026-09-09T00:00:00.000Z' }) - ] - const scores = matches(match(1, 10), match(2, 1)) - expect(rankSessionHits(sessions, scores, 'relevance').map((e) => e.session.id)).toEqual([1, 2]) - expect(rankSessionHits(sessions, scores, 'newest').map((e) => e.session.id)).toEqual([2, 1]) - }) - - it.each(['relevance', 'newest'] as const)( - 'breaks a %s tie by session, whatever order retrieval handed them over in', - (sort) => { - // A cursor is an offset into this list, so two entries that tie must not - // be free to swap between pages. Retrieval hands sessions over in - // whatever order the `IN (...)` lookup produced, which SQL does not - // promise, so the order below is deliberately reversed. - const sessions = [6, 5, 4, 3, 2, 1].map((id) => session(id)) - const scores = matches(...sessions.map((entry) => match(entry.id, 5))) - expect(rankSessionHits(sessions, scores, sort).map((entry) => entry.session.id)).toEqual([ - 1, 2, 3, 4, 5, 6 - ]) - } - ) - - it('prefers the shorter session when two match equally well', () => { - // The length prior: `0.02 · ln(1 + messages)`, subtracted per session. - const sessions = [session(1, { message_count: 5000 }), session(2, { message_count: 2 })] - const ranked = rankSessionHits(sessions, matches(match(1, 5), match(2, 5)), 'relevance') - expect(ranked.map((entry) => entry.session.id)).toEqual([2, 1]) - expect(ranked[0]!.score).toBeGreaterThan(ranked[1]!.score) - }) -}) - -describe('forks fold into one answer', () => { - const fork = (id: number, updatedAt: string): SessionRow => - session(id, { - updated_at: updatedAt, - content_hash: 'shared-opening-prefix', - content_hash_count: 8 - }) - - it('keeps the newest copy and counts the rest', () => { - const sessions = [ - fork(1, '2026-09-01T00:00:00.000Z'), - fork(2, '2026-09-09T00:00:00.000Z'), - fork(3, '2026-09-05T00:00:00.000Z') - ] - const ranked = rankSessionHits( - sessions, - matches(match(1, 9), match(2, 1), match(3, 5)), - 'relevance' - ) - expect(ranked).toHaveLength(1) - expect(ranked[0]!.session.id).toBe(2) - expect(ranked[0]!.duplicateCount).toBe(3) - }) - - it('leaves sessions with no shared prefix alone', () => { - const sessions = [session(1), session(2)] - const ranked = rankSessionHits(sessions, matches(match(1, 9), match(2, 5)), 'relevance') - expect(ranked.map((entry) => entry.duplicateCount)).toEqual([1, 1]) - }) -}) - -it('scores a session that matched no text at zero, less its length prior', () => { - // The operator-only page: there is no relevance signal, only an order. - const ranked = rankSessionHits([session(1, { message_count: 9 })], new Map(), 'newest') - expect(ranked[0]!.message).toBeNull() - expect(ranked[0]!.score).toBeLessThan(0) -}) diff --git a/src/main/ai-vault-search/session-search-hit-ranking.ts b/src/main/ai-vault-search/session-search-hit-ranking.ts deleted file mode 100644 index 364858ea650..00000000000 --- a/src/main/ai-vault-search/session-search-hit-ranking.ts +++ /dev/null @@ -1,109 +0,0 @@ -import type { AiVaultAgent } from '../../shared/ai-vault-types' -import { isCollapsibleContentHash } from './session-search-content-hash' -import type { SessionSearchSort } from './session-search-engine-types' - -// Subtracted per session: `0.02 · ln(1 + messages)`; slightly positive on both eval sets. -const LENGTH_PRIOR = 0.02 - -export type SessionRow = { - id: number - agent: AiVaultAgent - session_id: string - file_path: string - codex_home: string | null - title: string - cwd: string | null - branch: string | null - updated_at: string | null - message_count: number - resume_command: string - content_hash: string | null - content_hash_count: number -} - -/** The one message that stands for a session: its best-scoring match. */ -export type MessageRow = { - rowid: number - score: number - session_row_id: number - role: string - ts: string | null -} - -export type RankedSession = { - session: SessionRow - /** Null on an operator-only page: the session matched no text at all. */ - message: MessageRow | null - score: number - duplicateCount: number -} - -/** - * Everything between "these sessions matched" and "this is the ranked list": - * the length prior, fork folding and the caller's order. Retrieval stays in SQL - * and nothing here touches the database. - * - * The whole list is returned, not a page: a cursor indexes into it, and slicing - * here would make page two a different ranking from page one. The engine cuts - * the page and only then pays for a snippet. - */ -export function rankSessionHits( - sessions: readonly SessionRow[], - matches: ReadonlyMap, - sort: SessionSearchSort -): RankedSession[] { - const scored = collapseForks( - sessions.map((session) => { - const message = matches.get(session.id) ?? null - return { - session, - message, - score: (message?.score ?? 0) - LENGTH_PRIOR * Math.log(1 + session.message_count), - duplicateCount: 1 - } - }) - ) - // Why a total order and not just the key: a cursor is an offset into this - // list, so two entries that tie must not be free to swap between pages. - scored.sort( - (left, right) => - (sort === 'newest' - ? (right.session.updated_at ?? '').localeCompare(left.session.updated_at ?? '') - : right.score - left.score) || left.session.id - right.session.id - ) - return scored -} - -/** - * Folds forked copies of one conversation into a single entry: same opening - * prefix, newest `updated_at` wins, the rest become `duplicateCount`. Done here - * and not at write time so index rows stay per file (cursors and deletes). - */ -function collapseForks(scored: RankedSession[]): RankedSession[] { - const groups = new Map() - for (const entry of scored) { - const { content_hash: hash, content_hash_count: count, id } = entry.session - const key = isCollapsibleContentHash(hash, count) ? `hash:${hash}` : `session:${id}` - const group = groups.get(key) - if (group) { - group.push(entry) - } else { - groups.set(key, [entry]) - } - } - const collapsed: RankedSession[] = [] - for (const group of groups.values()) { - if (group.length === 1) { - collapsed.push(group[0]!) - continue - } - const winner = group.reduce((best, entry) => (isNewer(entry, best) ? entry : best)) - collapsed.push({ ...winner, duplicateCount: group.length }) - } - return collapsed -} - -function isNewer(entry: RankedSession, best: RankedSession): boolean { - const order = (entry.session.updated_at ?? '').localeCompare(best.session.updated_at ?? '') - return order === 0 ? entry.score > best.score : order > 0 -} diff --git a/src/main/ai-vault-search/session-search-index-generation.test.ts b/src/main/ai-vault-search/session-search-index-generation.test.ts deleted file mode 100644 index a3fffd2648f..00000000000 --- a/src/main/ai-vault-search/session-search-index-generation.test.ts +++ /dev/null @@ -1,320 +0,0 @@ -import { appendFile, mkdtemp, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, expect, it } from 'vitest' -import { removeTree } from '../../shared/windows-transient-lock-removal' -import type SyncDatabase from '../sqlite/sync-database' -import { SessionSearchEngine } from './session-search-engine' -import { readIndexGeneration } from './session-search-index-generation' -import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' -import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' -import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' -import type { SessionSearchCursorError } from './session-search-page-cursor' -import { openSessionSearchDatabase } from './session-search-schema' -import { SessionSearchStore } from './session-search-store' -import { parseTranscript, userRecord } from './session-search-transcript-fixtures' - -let roots: string[] = [] -let handles: SyncDatabase[] = [] - -afterEach(async () => { - resetTranscriptConsumersForTests() - resetSessionParseCacheForTests() - for (const handle of handles) { - handle.close() - } - handles = [] - await Promise.all(roots.map((root) => removeTree(root))) - roots = [] -}) - -async function tempRoot(): Promise { - const root = await mkdtemp(join(tmpdir(), 'orca-search-generation-')) - roots.push(root) - return root -} - -/** - * A reader's own handle on the index, with the engine's schema installed. - * - * PR 2's store keeps its connection private, so a reader opens its own — which - * is what the fence has to survive: nothing this handle does moves the - * generation, and it must still see every writer's move. - */ -function reader(path: string): SyncDatabase { - const db = openSessionSearchDatabase(path) - handles.push(db) - // Constructing an engine is what installs the triggers. - new SessionSearchEngine(db) - return db -} - -/** Indexes one transcript through the real consumer and returns its path. */ -async function indexOneTranscript(root: string, store: SessionSearchStore): Promise { - resetSessionParseCacheForTests() - const sessionId = `aaaaaaaa-0000-4000-8000-${String(roots.length).padStart(12, '0')}` - const path = join(root, `${Math.random().toString(36).slice(2)}.jsonl`) - await writeFile(path, `${userRecord(0, 'generation fixture needle', sessionId)}\n`) - const unregister = registerSessionSearchIndexConsumer(store) - try { - await parseTranscript(path) - } finally { - unregister() - } - return path -} - -it('moves the generation forward when a committed read changes what a read returns', async () => { - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - const before = readIndexGeneration(db) - await indexOneTranscript(root, store) - expect(readIndexGeneration(db)).toBeGreaterThan(before) - } finally { - store.close() - } -}) - -it('moves the generation forward when an append adds rows to a live session', async () => { - // The first read of a file inserts its `files` row; every read after that - // updates it. An append changes a session's rank and its message count, so a - // cursor minted before it indexes into a list that no longer exists. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - const transcript = await indexOneTranscript(root, store) - const indexed = readIndexGeneration(db) - const unregister = registerSessionSearchIndexConsumer(store) - try { - resetSessionParseCacheForTests() - await appendFile(transcript, `${userRecord(1, 'a second needle turn')}\n`) - await parseTranscript(transcript) - } finally { - unregister() - } - expect(db.prepare('SELECT COUNT(*) AS c FROM messages').get()).toEqual({ c: 2 }) - expect(readIndexGeneration(db)).toBeGreaterThan(indexed) - } finally { - store.close() - } -}) - -it('moves the generation forward when a proven deletion hides a session', async () => { - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - const transcript = await indexOneTranscript(root, store) - const indexed = readIndexGeneration(db) - store.removeFile(transcript) - expect(readIndexGeneration(db)).toBeGreaterThan(indexed) - } finally { - store.close() - } -}) - -it('moves the generation forward when retention cuts a session loose', async () => { - // Retention deletes the session row and the file row in one transaction, then - // reclaims the messages over many. It is the first half that changes what a - // search returns, and the first half that has to move the generation. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - await indexOneTranscript(root, store) - const indexed = readIndexGeneration(db) - await store.purgeOlderThan(Date.now() + 60_000) - expect(db.prepare('SELECT COUNT(*) AS c FROM sessions').get()).toEqual({ c: 0 }) - expect(readIndexGeneration(db)).toBeGreaterThan(indexed) - } finally { - store.close() - } -}) - -it('moves the generation when a purge reclaims rows nothing can reach', async () => { - // The drain writes only `messages`, and for a while that was argued to change - // no answer. Retrieval never saw those rows; the typo repair's dictionary - // did, because `messages_vocab` is a view over the FTS b-tree and lists a - // term whether or not a reader can reach it. See - // `session-search-orphan-rows.test.ts` for the answer that moved. The price - // of fencing it is a cursor refused once per batch while a purge runs. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - await indexOneTranscript(root, store) - // The shape an interrupted purge leaves: rows with no session row. - db.prepare('DELETE FROM sessions').run() - const orphaned = readIndexGeneration(db) - expect(db.prepare('SELECT COUNT(*) AS c FROM messages').get()).not.toEqual({ c: 0 }) - await store.purgeOlderThan(null) - expect(db.prepare('SELECT COUNT(*) AS c FROM messages').get()).toEqual({ c: 0 }) - expect(readIndexGeneration(db)).toBeGreaterThan(orphaned) - } finally { - store.close() - } -}) - -it("leaves the generation alone when a replace swaps a session's own rows", async () => { - // The same trigger must not fire here, or every re-read of a large transcript - // would move the generation once per deleted row on top of the one bump its - // file record already makes. A replace deletes rows whose session row still - // stands, which is what the trigger's `WHEN` clause tests. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - await indexOneTranscript(root, store) - const rows = db.prepare('SELECT COUNT(*) AS c FROM messages').get() as { c: number } - const indexed = readIndexGeneration(db) - db.prepare('DELETE FROM messages WHERE session_row_id IN (SELECT id FROM sessions)').run() - expect(rows.c).toBeGreaterThan(0) - expect(readIndexGeneration(db)).toBe(indexed) - } finally { - store.close() - } -}) - -it('leaves the generation alone when a removal hides nothing', async () => { - // A backfill retires paths it never held; if that moved the generation, every - // cursor would be refused for as long as indexing ran. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - await indexOneTranscript(root, store) - const before = readIndexGeneration(db) - store.removeFile('/synthetic/never-indexed.jsonl') - expect(readIndexGeneration(db)).toBe(before) - } finally { - store.close() - } -}) - -it('keeps the generation across a reopen, because the bump rides its own commit', async () => { - // The bump is inside the transaction that changes visibility, so nothing can - // be lost to a crash and reopening need not invalidate anyone's cursor. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - reader(path) - const first = new SessionSearchStore(path, (error) => { - throw error - }) - await indexOneTranscript(root, first) - const indexed = readIndexGeneration(reader(path)) - first.close() - - const second = new SessionSearchStore(path) - try { - expect(readIndexGeneration(reader(path))).toBe(indexed) - } finally { - second.close() - } -}) - -it('fences a reader against a writer it does not share a process with', async () => { - // The shape PR 3 creates: the indexer writes from the scanner child while an - // engine reads elsewhere. A generation cached in the reader's memory tracks - // only that reader's own writes, so it would stand still through the - // writer's deletion, honour the stale cursor, and skip a session. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const writer = new SessionSearchStore(path, (error) => { - throw error - }) - try { - const transcripts: string[] = [] - for (let n = 0; n < 3; n++) { - transcripts.push(await indexOneTranscript(root, writer)) - } - const engine = new SessionSearchEngine(db) - const page = engine.search({ query: 'needle', limit: 1 }) - expect(page.page.cursor).not.toBeNull() - - writer.removeFile(transcripts[0]!) - - // The reader never wrote anything, and must still refuse. - try { - engine.search({ query: 'needle', limit: 1, cursor: page.page.cursor! }) - expect.unreachable('a page cursor must not survive another writer moving the index') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('stale-generation') - } - } finally { - writer.close() - } -}) - -it('re-creates a fence something dropped, on the next search', async () => { - // An index whose triggers are gone cannot move its generation, so every stale - // cursor would compare equal and be honoured against a list the caller never - // saw. The engine owns those triggers, so it puts them back. - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const store = new SessionSearchStore(path, (error) => { - throw error - }) - try { - await indexOneTranscript(root, store) - const engine = new SessionSearchEngine(db) - db.exec('DROP TRIGGER search_generation_file_update') - engine.search({ query: 'needle' }) - - const restored = readIndexGeneration(db) - await indexOneTranscript(root, store) - expect(readIndexGeneration(db)).toBeGreaterThan(restored) - } finally { - store.close() - } -}) - -it('mints a distinct generation per change even when two handles write', async () => { - const root = await tempRoot() - const path = join(root, 'index.sqlite') - const db = reader(path) - const first = new SessionSearchStore(path, (error) => { - throw error - }) - const second = new SessionSearchStore(path, (error) => { - throw error - }) - try { - const seen: number[] = [readIndexGeneration(db)] - for (const store of [first, second, first, second]) { - await indexOneTranscript(root, store) - seen.push(readIndexGeneration(db)) - } - // Read-then-write from two connections would hand out one value twice. - expect(new Set(seen).size).toBe(seen.length) - expect([...seen].sort((left, right) => left - right)).toEqual(seen) - } finally { - second.close() - first.close() - } -}) diff --git a/src/main/ai-vault-search/session-search-index-generation.ts b/src/main/ai-vault-search/session-search-index-generation.ts deleted file mode 100644 index 891608ed2e7..00000000000 --- a/src/main/ai-vault-search/session-search-index-generation.ts +++ /dev/null @@ -1,97 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' - -const GENERATION_KEY = 'index_generation' - -/** - * Names of the triggers that move the generation. Exported so the query schema - * can check they are all still there before an engine trusts a cursor. - */ -export const SESSION_SEARCH_GENERATION_TRIGGERS = [ - 'search_generation_file_insert', - 'search_generation_file_update', - 'search_generation_file_delete', - 'search_generation_orphan_reclaim' -] as const - -const BUMP = `INSERT INTO meta(key, value) VALUES ('${GENERATION_KEY}', '1') - ON CONFLICT(key) DO UPDATE SET value = CAST(value AS INTEGER) + 1;` - -/** - * The fence, as three triggers on `files`. - * - * Why `files`. Every transaction the store opens that can change what a search - * returns writes this table: a committed read upserts the file's cursor beside - * its rows, a chunk of a long read upserts the partial sentinel beside its - * prefix, `removeFile` deletes the row with the session, and retention deletes - * the file row in the same transaction as the session row. - * - * And why `messages` as well, for orphans only. Retention's second half - * reclaims rows whose session row is already gone, and touches neither table - * above. It was left unfenced on the argument that those rows answer nothing, - * which is true of retrieval and was not true of the whole engine: the typo - * repair's dictionary is `messages_vocab`, a view over the FTS b-tree that - * lists a term whether or not a reader can reach the rows carrying it, and - * reclaiming them moved which word a query was repaired to. The repair now - * counts live rows instead, so the common case is fixed at its source; this - * trigger is what makes the fence true rather than nearly true, because the - * vocabulary still decides which candidates survive its scan limit. - * - * The `WHEN` clause is what keeps it free. `removeFile` deletes a session's - * rows while its `sessions` row still stands, so it does not fire here; a - * replace cuts the old `sessions` row loose and leaves its messages to the - * drain (PR 2 round 10). Both already bump through `files`; the drain is the - * only path that deletes a row whose session is gone, and it fires here. The cost of the fence is real and worth naming: a - * cursor outstanding while a purge runs is refused once per batch, which - * `SessionSearchCursorError` reports as `stale-generation` so a caller - * re-issues page one rather than showing anyone an error. - * - * A trigger rather than a call the writer makes, for two reasons. PR 4 does not - * own the writer, and more importantly the fence has to hold for writers this - * process cannot see: the triggers live in the file, so PR 3's indexer in the - * scanner child moves the generation without knowing a reader exists. - * - * Correctness comes from where the increment runs, not from what it counts. It - * is one statement inside the writer's own `BEGIN IMMEDIATE`, so it commits - * with the change it describes and two connections cannot mint one value twice. - * It over-counts in one harmless direction: a read that decoded no session from - * a file the index also held no session for advances a cursor and bumps - * anyway. That refuses a cursor early; it never honours one late. - */ -export const SESSION_SEARCH_GENERATION_SQL = ` -CREATE TRIGGER IF NOT EXISTS search_generation_file_insert AFTER INSERT ON files BEGIN - ${BUMP} -END; -CREATE TRIGGER IF NOT EXISTS search_generation_file_update AFTER UPDATE ON files BEGIN - ${BUMP} -END; -CREATE TRIGGER IF NOT EXISTS search_generation_file_delete AFTER DELETE ON files BEGIN - ${BUMP} -END; -CREATE TRIGGER IF NOT EXISTS search_generation_orphan_reclaim AFTER DELETE ON messages -WHEN NOT EXISTS (SELECT 1 FROM sessions WHERE id = OLD.session_row_id) BEGIN - ${BUMP} -END; -` - -/** - * A monotone id for what the index currently publishes. - * - * A search page is a slice of one ranked list, so a cursor only means anything - * against the snapshot that produced it. Every change to what a read can return - * moves this on, and a cursor minted under an older value is refused rather - * than silently re-run against a list it no longer indexes into. - * - * Read from the database on every call, never cached in a process. The writer - * and the reader need not be the same one: PR 3's indexer runs in the scanner - * child while an engine reads elsewhere, and any number of handles may be open - * on one file. A generation cached in memory only ever tracks that process's - * own writes, so a reader would see another writer's deletions while its - * generation stood still, honour a stale cursor, and skip a session. - */ -export function readIndexGeneration(db: SyncDatabase): number { - const row = db.prepare('SELECT value FROM meta WHERE key = ?').get(GENERATION_KEY) as - | { value: string } - | undefined - const parsed = row ? Number(row.value) : Number.NaN - return Number.isInteger(parsed) && parsed >= 0 ? parsed : 0 -} diff --git a/src/main/ai-vault-search/session-search-orphan-rows.test.ts b/src/main/ai-vault-search/session-search-orphan-rows.test.ts deleted file mode 100644 index dfda2303104..00000000000 --- a/src/main/ai-vault-search/session-search-orphan-rows.test.ts +++ /dev/null @@ -1,186 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import type SyncDatabase from '../sqlite/sync-database' -import { - addSyntheticSession, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' -import { identifierShadowText } from './session-search-identifier-split' -import { readIndexGeneration } from './session-search-index-generation' -import { planSessionSearchQuery } from './session-search-query-planner' -import { sessionSearchSnippet } from './session-search-snippet' -import type { SessionSearchCursorError } from './session-search-page-cursor' -import { SessionSearchTypoRepair } from './session-search-typo-repair' - -// Retention deletes a session row in one small transaction and reclaims its -// message rows in batches afterwards, so a `messages` row with no `sessions` row -// is a state every purge, every removed source and every interrupted drain -// passes through. Those rows are still in both FTS tables and still in the -// vocabulary, and nothing here may return one. -// -// A hit is a session row, and the ranked list is loaded `FROM sessions`, so the -// route ladder below cannot surface an orphan even if a join were loosened — -// those cases are a ratchet over the shape, not the proof. The two reads that -// can leak one are pinned separately and each is a real oracle: the snippet, -// which is handed a rowid and asked for its text, and the typo repair, whose -// dictionary is the FTS b-tree and lists an orphan's terms like any other. - -const ORPHAN_SESSION_ROW = 99 -const ORPHAN_TEXT = 'orphaned marmoset secret' - -let harness: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - harness = null -}) - -/** Two rows in the FTS table and the vocabulary, and no session row for them. */ -function plantOrphans(db: SyncDatabase, text: string = ORPHAN_TEXT): number[] { - const rowids: number[] = [] - for (let n = 0; n < 2; n++) { - const rowid = Number( - db - .prepare("INSERT INTO messages(session_row_id,role,ts) VALUES (?,'user',?)") - .run(ORPHAN_SESSION_ROW, '2026-09-10T00:00:00.000Z').lastInsertRowid - ) - db.prepare( - 'INSERT INTO messages_fts(rowid,user_text,assistant_text,tool_text,identifiers) VALUES (?,?,?,?,?)' - ).run(rowid, text, '', '', identifierShadowText(text)) - rowids.push(rowid) - } - return rowids -} - -async function withOrphans(): Promise<{ harness: SessionSearchHarness; rowids: number[] }> { - harness = await openSessionSearchHarness('ss-orphan-rows') - addSyntheticSession(harness.db, { id: 1, text: 'the haystack line here' }) - const rowids = plantOrphans(harness.db) - // The oracle only means anything if the rows are really there to be found. - expect( - harness.db - .prepare("SELECT count(*) AS c FROM messages_fts WHERE messages_fts MATCH 'marmoset'") - .get() - ).toEqual({ c: 2 }) - expect( - harness.db.prepare("SELECT doc FROM messages_vocab WHERE term = 'marmoset'").get() - ).toEqual({ doc: 2 }) - return { harness, rowids } -} - -it.each([ - ['phrase', '"orphaned marmoset"'], - ['and', 'orphaned secret'], - ['single-token literal', 'marmoset'], - ['or', 'marmoset haystack orphaned'], - ['typo repair', 'marmosett'], - ['operator only', 'repo:app'] -])('returns no orphaned row on the %s route', async (_route, query) => { - const { harness: open } = await withOrphans() - for (const scope of ['all', 'conversation'] as const) { - const hits = open.engine.search({ query, scope }).hits - expect(hits.map((hit) => hit.sessionId)).not.toContain(String(ORPHAN_SESSION_ROW)) - expect(hits.filter((hit) => hit.evidence?.snippet.includes('marmoset'))).toEqual([]) - } -}) - -it('never repairs a term onto a spelling only orphaned rows carry', async () => { - const { harness: open } = await withOrphans() - // `marmoset` is in the vocabulary twice, which is what would make it the - // repair for `marmosett` if the repair trusted the vocabulary alone. - expect(new SessionSearchTypoRepair(open.db).correct('marmosett', 'all')).toBeNull() - expect(open.engine.search({ query: 'marmosett' }).planner.repairedTerms).toBeUndefined() -}) - -it('snippets nothing for an orphaned row, even asked for it by rowid', async () => { - const { harness: open, rowids } = await withOrphans() - const plan = planSessionSearchQuery('marmoset') - for (const scope of ['all', 'conversation'] as const) { - expect(sessionSearchSnippet(open.db, scope, rowids[0]!, plan)).toEqual({ - text: '', - truncated: false - }) - } -}) - -it('still answers for the live session beside them', async () => { - const { harness: open } = await withOrphans() - expect(open.engine.search({ query: 'haystack' }).hits.map((hit) => hit.sessionId)).toEqual(['1']) -}) - -// Reclaiming those rows is the other half. The drain deletes only from -// `messages`, so for a long time it was argued to change no answer and left -// outside the generation fence. Retrieval never saw them, but the typo repair's -// dictionary is `messages_vocab`, a view over the FTS b-tree that lists a term -// whether or not a reader can reach the rows carrying it — so the drain moved -// which word a query was repaired to, under a cursor that was still honoured. -describe('a purge reclaiming rows nothing can reach', () => { - /** A live session and a purged one that both carry `text`. */ - async function withReclaimable(): Promise { - harness = await openSessionSearchHarness('ss-orphan-drain') - // Two live rows, which is what makes `marmoset` eligible as a repair at all. - addSyntheticSession(harness.db, { id: 1, text: 'the marmoset lives here', rows: 2 }) - plantOrphans(harness.db) - return harness - } - - it('answers the same before and after, because the repair counts live rows', async () => { - const open = await withReclaimable() - const before = open.engine.search({ query: 'marmosett' }) - expect(before.planner.repairedTerms).toEqual(['marmoset']) - expect(before.hits.map((hit) => hit.sessionId)).toEqual(['1']) - - await open.store.purgeOlderThan(null) - expect(open.db.prepare('SELECT count(*) AS c FROM messages').get()).toEqual({ c: 2 }) - - const after = open.engine.search({ query: 'marmosett' }) - expect(after.planner.repairedTerms).toEqual(before.planner.repairedTerms) - expect(after.hits.map((hit) => hit.sessionId)).toEqual(before.hits.map((hit) => hit.sessionId)) - }) - - it('moves the generation anyway, so no cursor spans it', async () => { - // The repair counting live rows fixes the common case. It does not make the - // drain provably inert: `messages_vocab` still decides which candidates - // survive its scan limit, and reclaiming a term's last row changes where - // that limit cuts. The fence is what covers the rest, at the price of - // refusing a cursor once per batch while a purge runs. - const open = await withReclaimable() - // A second live session, so page one has a page two to be refused. - addSyntheticSession(open.db, { id: 2, text: 'the marmoset again', rows: 2 }) - const page = open.engine.search({ query: 'marmoset', limit: 1 }) - expect(page.page.cursor).not.toBeNull() - const before = readIndexGeneration(open.db) - - await open.store.purgeOlderThan(null) - - expect(readIndexGeneration(open.db)).toBeGreaterThan(before) - try { - open.engine.search({ query: 'marmoset', limit: 1, cursor: page.page.cursor! }) - expect.unreachable('a cursor must not span a purge') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('stale-generation') - } - }) - - it('picks the same repair when an unreachable spelling was the more common one', async () => { - // Two candidates equally close to the query. `marmosetx` led on the old - // ranking only because two of its rows belonged to a session retention had - // already cut loose, so the drain swapped the repair under a live cursor. - harness = await openSessionSearchHarness('ss-orphan-drain-tie') - const db = harness.db - for (let id = 1; id <= 4; id++) { - addSyntheticSession(db, { id, text: `marmosetx session${id}` }) - } - for (let id = 5; id <= 9; id++) { - addSyntheticSession(db, { id, text: `marmosetq session${id}` }) - } - plantOrphans(db, 'marmosetx') - - const before = harness.engine.search({ query: 'marmosett' }) - expect(before.planner.repairedTerms).toEqual(['marmosetq']) - await harness.store.purgeOlderThan(null) - expect(harness.engine.search({ query: 'marmosett' }).planner.repairedTerms).toEqual( - before.planner.repairedTerms - ) - }) -}) diff --git a/src/main/ai-vault-search/session-search-page-cursor.ts b/src/main/ai-vault-search/session-search-page-cursor.ts deleted file mode 100644 index 838e5e1e3ab..00000000000 --- a/src/main/ai-vault-search/session-search-page-cursor.ts +++ /dev/null @@ -1,108 +0,0 @@ -import { createHash } from 'node:crypto' -import type { SessionSearchRequest } from './session-search-engine-types' - -export type SessionSearchCursorRejection = 'stale-generation' | 'different-query' | 'malformed' - -/** - * A cursor the engine refuses to honour. Typed, and thrown rather than - * swallowed: silently restarting at page one hands the caller a page it has - * already shown as if it were the next one, and silently re-running against a - * newer index hands it a slice of a list it never saw. - */ -export class SessionSearchCursorError extends Error { - constructor( - readonly rejection: SessionSearchCursorRejection, - /** - * The generation the index is at now. Always present: the engine knows it - * before it looks at the cursor at all. - */ - readonly actualGeneration: number, - /** - * The generation the cursor claims it was minted in. Absent only when the - * cursor could not be decoded far enough to carry a number, which is one of - * the `malformed` cases. - */ - readonly expectedGeneration?: number - ) { - super(`Search cursor rejected: ${rejection}`) - this.name = 'SessionSearchCursorError' - } -} - -type CursorPayload = { - /** Index generation. */ - g: number - /** - * Offset into the ranked list, not a session id. Ids are not in a cursor at - * all, so nothing here depends on `sessions.id` being unique over time — - * though it is, because PR 2 made the column AUTOINCREMENT so a purged - * session's id is never reissued to a live one. - */ - o: number - /** Query identity; see `sessionSearchPageKey`. */ - k: string -} - -/** - * Everything a page's ranking depends on except the limit. Two requests with - * the same key produce the same ranked list within one generation, so a cursor - * minted by one is meaningful to the other; the limit is left out on purpose so - * a caller may change its page size mid-pagination. - */ -export function sessionSearchPageKey(request: SessionSearchRequest): string { - const filters = request.filters ?? {} - const identity = JSON.stringify([ - request.query, - request.scope ?? 'all', - filters.sort ?? 'relevance', - filters.since ?? null, - [...(filters.agents ?? [])].sort(), - [...(filters.scopePaths ?? [])].sort() - ]) - return createHash('sha256').update(identity).digest('base64url').slice(0, 16) -} - -export function encodeSessionSearchCursor(generation: number, offset: number, key: string): string { - const payload: CursorPayload = { g: generation, o: offset, k: key } - return Buffer.from(JSON.stringify(payload), 'utf-8').toString('base64url') -} - -/** - * The offset this cursor points at, or a typed rejection. - * - * Every rejection carries `actualGeneration`, and every one that could read a - * generation out of the cursor carries `expectedGeneration` too, so a caller - * can tell "the index moved under you, ask for page one" from "this cursor is - * not ours" and act on the first without showing anyone an error. - */ -export function decodeSessionSearchCursor(cursor: string, generation: number, key: string): number { - let payload: CursorPayload - try { - payload = JSON.parse(Buffer.from(cursor, 'base64url').toString('utf-8')) as CursorPayload - } catch { - throw new SessionSearchCursorError('malformed', generation) - } - // A generation that survived parsing is worth reporting even when the rest of - // the payload is unusable: it is what tells the caller which snapshot the - // cursor thought it was walking. - const claimed = - typeof payload?.g === 'number' && Number.isFinite(payload.g) ? payload.g : undefined - if ( - claimed === undefined || - !Number.isInteger(payload?.o) || - payload.o < 0 || - typeof payload?.k !== 'string' - ) { - throw new SessionSearchCursorError('malformed', generation, claimed) - } - // Generation first: a caller who changed the query AND waited through a - // publish should hear about the index moving, which is the condition it - // cannot fix by paging again. - if (claimed !== generation) { - throw new SessionSearchCursorError('stale-generation', generation, claimed) - } - if (payload.k !== key) { - throw new SessionSearchCursorError('different-query', generation, claimed) - } - return payload.o -} diff --git a/src/main/ai-vault-search/session-search-paging.test.ts b/src/main/ai-vault-search/session-search-paging.test.ts deleted file mode 100644 index 319c5b0dba4..00000000000 --- a/src/main/ai-vault-search/session-search-paging.test.ts +++ /dev/null @@ -1,309 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import type { SessionSearchRequest } from './session-search-engine-types' -import { - addSyntheticSession, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' -import { readIndexGeneration } from './session-search-index-generation' -import { - decodeSessionSearchCursor, - encodeSessionSearchCursor, - SessionSearchCursorError, - sessionSearchPageKey -} from './session-search-page-cursor' - -let harness: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - harness = null -}) - -async function open(name: string, options = {}): Promise { - harness = await openSessionSearchHarness(name, options) - return harness -} - -async function withSessions(count: number, options = {}): Promise { - harness = await openSessionSearchHarness('ss-engine-paging', options) - for (let id = 1; id <= count; id++) { - addSyntheticSession(harness.db, { - id, - text: `needle padding ${'word '.repeat(id % 5)}`, - updatedAt: `2026-09-${String(id).padStart(2, '0')}T00:00:00.000Z` - }) - } - return harness -} - -describe('a cursor walks one ranked list', () => { - it('pages through every session exactly once, in one stable order', async () => { - const { engine } = await withSessions(25) - const request: SessionSearchRequest = { query: 'needle', limit: 10 } - const seen: string[] = [] - let cursor: string | null = null - let pages = 0 - do { - const page = engine.search(cursor ? { ...request, cursor } : request) - seen.push(...page.hits.map((hit) => hit.sessionId)) - cursor = page.page.cursor - pages++ - expect(pages).toBeLessThan(10) - } while (cursor !== null) - - expect(pages).toBe(3) - expect(seen).toHaveLength(25) - expect(new Set(seen).size).toBe(25) - // The same walk, run again against the same generation, is the same walk. - expect(engine.search(request).hits.map((hit) => hit.sessionId)).toEqual(seen.slice(0, 10)) - }) - - it('closes the page when the last hit has been handed out', async () => { - const { engine } = await withSessions(3) - const page = engine.search({ query: 'needle', limit: 10 }) - expect(page.hits).toHaveLength(3) - expect(page.page.hasMore).toBe(false) - expect(page.page.cursor).toBeNull() - }) - - it('lets a caller change page size mid-walk', async () => { - const { engine } = await withSessions(12) - const first = engine.search({ query: 'needle', limit: 5 }) - const rest = engine.search({ query: 'needle', limit: 20, cursor: first.page.cursor! }) - expect(rest.hits).toHaveLength(7) - expect(rest.page.hasMore).toBe(false) - }) - - it('breaks a tie by session, so two entries cannot swap between pages', async () => { - // Same text, same timestamp: every ranking key is equal, which is exactly - // where an unstable sort would hand one session out twice and lose another. - harness = await openSessionSearchHarness('ss-engine-ties') - for (let id = 1; id <= 6; id++) { - addSyntheticSession(harness.db, { id, text: 'needle', updatedAt: '2026-09-01T00:00:00.000Z' }) - } - const first = harness.engine.search({ query: 'needle', limit: 3 }) - const second = harness.engine.search({ query: 'needle', limit: 3, cursor: first.page.cursor! }) - const seen = [...first.hits, ...second.hits].map((hit) => hit.sessionId) - expect(seen).toEqual(['1', '2', '3', '4', '5', '6']) - }) -}) - -describe('a cursor is refused rather than reinterpreted', () => { - it('rejects a cursor minted before the index moved', async () => { - const { engine, store } = await withSessions(25) - const first = engine.search({ query: 'needle', limit: 10 }) - // A proven deletion of a path this index really held hides a session, which - // is exactly the change a cursor must not be allowed to page across. - store.removeFile('/synthetic/1.jsonl') - - expect(() => engine.search({ query: 'needle', limit: 10, cursor: first.page.cursor! })).toThrow( - SessionSearchCursorError - ) - try { - engine.search({ query: 'needle', limit: 10, cursor: first.page.cursor! }) - expect.unreachable('a stale cursor must not be silently re-run') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('stale-generation') - } - }) - - it('names both generations, so a caller can tell a moved index from a bad cursor', async () => { - // What a caller does about it differs: a moved index means quietly ask for - // page one again, a bad cursor means something is wrong with the caller. - const { engine, store } = await withSessions(25) - const first = engine.search({ query: 'needle', limit: 10 }) - const minted = readIndexGeneration(harness!.db) - // Any published read moves the generation, including one for a file this - // page never mentioned. That is the fence working, not a defect. - store.removeFile('/synthetic/9.jsonl') - - try { - engine.search({ query: 'needle', limit: 10, cursor: first.page.cursor! }) - expect.unreachable('the index moved') - } catch (error) { - const rejected = error as SessionSearchCursorError - expect(rejected.rejection).toBe('stale-generation') - expect(rejected.expectedGeneration).toBe(minted) - expect(rejected.actualGeneration).toBe(readIndexGeneration(harness!.db)) - expect(rejected.actualGeneration).toBeGreaterThan(rejected.expectedGeneration!) - } - }) - - it('rejects a cursor carried over to a different query', async () => { - const { engine } = await withSessions(25) - const first = engine.search({ query: 'needle', limit: 10 }) - try { - engine.search({ query: 'padding', limit: 10, cursor: first.page.cursor! }) - expect.unreachable('a cursor indexes into one ranked list, not any list') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('different-query') - } - }) - - it('rejects a cursor whose filters changed, which reranks the list', async () => { - const { engine } = await withSessions(25) - const first = engine.search({ query: 'needle', limit: 10 }) - try { - engine.search({ - query: 'needle', - limit: 10, - cursor: first.page.cursor!, - filters: { sort: 'newest' } - }) - expect.unreachable('a different sort is a different ranked list') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('different-query') - } - }) - - // Every field the ranked list depends on has to be in the key, and a field - // that is in the key but never pinned is a field a refactor can drop while - // the suite stays green. One case each, through the engine, so the assertion - // is about a refused page and not about a hash. - it.each([ - ['scope', { scope: 'conversation' as const }], - ['sort', { filters: { sort: 'newest' as const } }], - ['agents', { filters: { agents: ['codex' as const] } }], - ['scopePaths', { filters: { scopePaths: ['/repo/app'] } }], - ['since', { filters: { since: '2026-09-01T00:00:00.000Z' } }] - ])('rejects a cursor presented with a different %s', async (_field, changed) => { - const { engine } = await withSessions(25) - const request: SessionSearchRequest = { - query: 'needle', - limit: 10, - scope: 'all', - filters: { sort: 'relevance', agents: ['claude'], scopePaths: ['/'], since: undefined } - } - const first = engine.search(request) - expect(first.page.cursor).not.toBeNull() - try { - engine.search({ - ...request, - ...changed, - filters: { ...request.filters, ...('filters' in changed ? changed.filters : {}) }, - cursor: first.page.cursor! - }) - expect.unreachable('a narrowing the ranked list depends on must invalidate the cursor') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('different-query') - } - }) - - it('rejects a cursor that is not one of ours', async () => { - const { engine } = await withSessions(3) - try { - engine.search({ query: 'needle', cursor: 'not-a-cursor' }) - expect.unreachable('a malformed cursor is not an empty one') - } catch (error) { - expect((error as SessionSearchCursorError).rejection).toBe('malformed') - } - }) -}) - -describe('cursor encoding', () => { - const request: SessionSearchRequest = { query: 'needle', filters: { scopePaths: ['/a'] } } - - it('round-trips an offset within its own generation and query', () => { - const key = sessionSearchPageKey(request) - expect(decodeSessionSearchCursor(encodeSessionSearchCursor(7, 40, key), 7, key)).toBe(40) - }) - - it('keys a request by what changes its ranking, and not by its page size', () => { - expect(sessionSearchPageKey({ ...request, limit: 5 })).toBe( - sessionSearchPageKey({ ...request, limit: 50 }) - ) - expect(sessionSearchPageKey({ ...request, scope: 'conversation' })).not.toBe( - sessionSearchPageKey(request) - ) - }) - - it('reads a filter list in any order as the same request', () => { - expect(sessionSearchPageKey({ query: 'a', filters: { agents: ['claude', 'codex'] } })).toBe( - sessionSearchPageKey({ query: 'a', filters: { agents: ['codex', 'claude'] } }) - ) - }) - - it.each([ - ['a negative offset', encodeSessionSearchCursor(1, -1, 'k'), 1], - ['a non-integer offset', Buffer.from('{"g":1,"o":1.5,"k":"k"}').toString('base64url'), 1], - ['a payload that is not an object', Buffer.from('"nope"').toString('base64url'), undefined], - ['text that is not base64url JSON', 'zzz!!', undefined] - ])('rejects %s as malformed, still naming the index generation', (_name, cursor, claimed) => { - // The caller has to know which snapshot it was refused against whatever was - // wrong with the cursor, and the generation it claimed whenever that - // survived parsing. - try { - decodeSessionSearchCursor(cursor, 7, 'k') - expect.unreachable('a malformed cursor is not an empty one') - } catch (error) { - const rejected = error as SessionSearchCursorError - expect(rejected.rejection).toBe('malformed') - expect(rejected.actualGeneration).toBe(7) - expect(rejected.expectedGeneration).toBe(claimed) - } - }) -}) - -describe('the candidate limit is a tunable default, and says when it cut', () => { - it('does not claim truncation when every session fits', async () => { - const { engine } = await withSessions(5, { sessionCandidateLimit: 600 }) - expect(engine.search({ query: 'needle' }).truncated.candidates).toBe(false) - }) - - it('claims truncation, and ranks only what it retrieved, at the limit', async () => { - const { engine } = await withSessions(10, { sessionCandidateLimit: 4 }) - const result = engine.search({ query: 'needle', limit: 100 }) - expect(result.truncated.candidates).toBe(true) - expect(result.hits).toHaveLength(4) - }) - - it('applies the same limit to an operator-only page', async () => { - const { engine } = await withSessions(10, { sessionCandidateLimit: 4 }) - const result = engine.search({ query: 'repo:app', limit: 100 }) - expect(result.truncated.candidates).toBe(true) - expect(result.hits).toHaveLength(4) - }) - - it('says it gave up when the operator walk stopped scanning, not that it is done', async () => { - // The shape that reads as a confident empty answer: the only match sits - // past the walk's ceiling, so the walk stops having found nothing. Zero - // hits and `truncated.candidates` false would tell a caller there is - // nothing to find, which is a different claim from "I stopped looking". - // The walk reads a page at a time and gives up past a ceiling of - // `candidateLimit` x 20, so the corpus has to be deeper than one page for - // the ceiling to be what ends it. The only match is the oldest session. - const deep = 600 - const { db, engine } = await open('ss-engine-sparse-deep', { sessionCandidateLimit: 2 }) - for (let id = 1; id <= deep; id++) { - addSyntheticSession(db, { - id, - cwd: id === deep ? '/repo/needleonly' : '/repo/app', - updatedAt: new Date(Date.UTC(2026, 8, 9) - id * 60_000).toISOString() - }) - } - const result = engine.search({ query: 'repo:needleonly' }) - expect(result.hits).toHaveLength(0) - expect(result.truncated.candidates).toBe(true) - }) - - it('does not claim it gave up when the walk really did read everything', async () => { - const { db, engine } = await open('ss-engine-sparse-shallow', { sessionCandidateLimit: 600 }) - addSyntheticSession(db, { id: 1, cwd: '/repo/app' }) - const result = engine.search({ query: 'repo:nothing-here' }) - expect(result.hits).toHaveLength(0) - expect(result.truncated.candidates).toBe(false) - }) -}) - -describe('the response carries the snapshot it was built from', () => { - it('reports the index generation on every result', async () => { - const { db, engine, store } = await withSessions(3) - const before = engine.search({ query: 'needle' }).generation - expect(before).toBe(readIndexGeneration(db)) - store.removeFile('/synthetic/1.jsonl') - const after = engine.search({ query: 'needle' }).generation - expect(after).toBe(readIndexGeneration(db)) - expect(after).toBeGreaterThan(before) - }) -}) diff --git a/src/main/ai-vault-search/session-search-query-log.test.ts b/src/main/ai-vault-search/session-search-query-log.test.ts deleted file mode 100644 index 7b88e628a83..00000000000 --- a/src/main/ai-vault-search/session-search-query-log.test.ts +++ /dev/null @@ -1,61 +0,0 @@ -import { afterEach, expect, it } from 'vitest' -import { - addSyntheticSession, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' -import { logSessionSearchQuery } from './session-search-query-log' - -let harness: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - harness = null -}) - -async function open(options = {}): Promise { - harness = await openSessionSearchHarness('ss-query-log', options) - addSyntheticSession(harness.db, { id: 1, text: 'needle' }) - return harness -} - -function loggedQueries(harness: SessionSearchHarness): string[] { - return ( - harness.db.prepare('SELECT query FROM search_log ORDER BY id').all() as { query: string }[] - ).map((row) => row.query) -} - -it('writes nothing on the query path unless the caller asked for a log', async () => { - const opened = await open() - opened.engine.search({ query: 'needle' }) - expect(loggedQueries(opened)).toEqual([]) -}) - -it('records the query and its route when logging is on', async () => { - const opened = await open({ logQueries: true }) - opened.engine.search({ query: 'needle' }) - const rows = opened.db.prepare('SELECT query, route, hits FROM search_log').all() as { - query: string - route: string - hits: number - }[] - expect(rows).toEqual([{ query: 'needle', route: 'or', hits: 1 }]) -}) - -it('stores the query as typed, the way the index stores content as written', async () => { - // PR 2 decided the index does not redact: it is a second copy of plaintext - // the user already holds under their own home directory. The same holds for - // what they typed into the search box. - const opened = await open({ logQueries: true }) - opened.engine.search({ query: 'Bearer abcdefghijklmnopqrstuvwxyz012345' }) - expect(loggedQueries(opened)[0]).toBe('Bearer abcdefghijklmnopqrstuvwxyz012345') -}) - -it('keeps the newest N and drops the rest, so the log cannot grow with use', async () => { - const opened = await open() - // The real ceiling is 5,000; the trim is the same statement at any size. - for (let n = 0; n < 12; n++) { - logSessionSearchQuery(opened.db, { query: `q${n}`, route: 'or', hits: 0, durationMs: 1 }, 5) - } - expect(loggedQueries(opened)).toEqual(['q7', 'q8', 'q9', 'q10', 'q11']) -}) diff --git a/src/main/ai-vault-search/session-search-query-log.ts b/src/main/ai-vault-search/session-search-query-log.ts deleted file mode 100644 index cdc5ce54425..00000000000 --- a/src/main/ai-vault-search/session-search-query-log.ts +++ /dev/null @@ -1,30 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' - -export const SEARCH_LOG_LIMIT = 5000 - -/** - * Local-only telemetry the eval set is rebuilt from. - * - * The query is stored as typed, for the reason PR 2 gives for not redacting - * transcript content: this file sits beside an index that already holds the - * user's own plaintext, so a second copy of what they typed is not a new - * exposure. What may leave the machine is a transport policy and belongs where - * the wire is. - * - * Nothing enables this by default: the engine writes a row only when its caller - * asked for it, because a log write on the query path is a write on what is - * otherwise a read-only lane. Who turns it on is PR 3b's settings decision. - */ -export function logSessionSearchQuery( - db: SyncDatabase, - entry: { query: string; route: string; hits: number; durationMs: number }, - limit: number = SEARCH_LOG_LIMIT -): void { - db.prepare( - 'INSERT INTO search_log(ts, query, route, hits, duration_ms) VALUES (?, ?, ?, ?, ?)' - ).run(new Date().toISOString(), entry.query, entry.route, entry.hits, entry.durationMs) - db.prepare( - `DELETE FROM search_log WHERE id <= ( - SELECT id FROM search_log ORDER BY id DESC LIMIT 1 OFFSET ?)` - ).run(limit) -} diff --git a/src/main/ai-vault-search/session-search-query-planner.test.ts b/src/main/ai-vault-search/session-search-query-planner.test.ts deleted file mode 100644 index ac4874b1d10..00000000000 --- a/src/main/ai-vault-search/session-search-query-planner.test.ts +++ /dev/null @@ -1,84 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { - andExpression, - isLiteralQuery, - orExpression, - phraseExpression, - planSessionSearchQuery, - quoteFtsTerm -} from './session-search-query-planner' - -describe('literal shape decides whether the phrase route is even tried', () => { - it.each([ - 'resolveTerminalPath', - 'src/main/foo-bar.ts', - 'MAX_RETRY_COUNT', - 'kern.tty.ptmx_max', - '#19687', - 'STA-4850', - '"exact words here"', - 'TypeError: undefined', - 'foo() {' - ])('treats %s as quoting something from a transcript', (query) => { - expect(isLiteralQuery(query)).toBe(true) - }) - - it.each(['why is the terminal slow', 'how do I resume a session', 'relay capacity'])( - 'treats %s as prose', - (query) => { - expect(isLiteralQuery(query)).toBe(false) - } - ) -}) - -describe('the body is what the phrase and AND routes see', () => { - it('drops stop words from prose so the AND route is not defeated by "the"', () => { - expect(planSessionSearchQuery('why is the relay dropping frames').body).toEqual([ - 'relay', - 'dropping', - 'frames' - ]) - }) - - it('keeps stop words inside a literal, where they are part of what was quoted', () => { - // The literal shape is `foo.ts`; dropping `the` would change what was typed. - expect(planSessionSearchQuery('the foo.ts file').body).toEqual(['the', 'foo.ts', 'file']) - }) - - it('keeps a query that is nothing but stop words rather than answering nothing', () => { - expect(planSessionSearchQuery('how do I').body).toEqual(['how', 'do', 'I']) - }) - - it('has no terms for a query with no searchable token', () => { - expect(planSessionSearchQuery(' ... ').terms).toEqual([]) - }) -}) - -describe('the OR fallback fans an identifier out into its pieces', () => { - it('adds the split pieces after the whole term, never in place of it', () => { - const plan = planSessionSearchQuery('resolveTerminalPath') - expect(plan.terms[0]).toBe('resolveTerminalPath') - expect(plan.terms).toContain('terminal') - expect(plan.terms).toContain('path') - // `resolve` is not a stop word, so the whole identifier is reachable by piece. - expect(plan.terms).toContain('resolve') - }) - - it('leaves an ordinary word alone', () => { - expect(planSessionSearchQuery('relay').terms).toEqual(['relay']) - }) -}) - -describe('FTS5 expressions quote every term', () => { - it('quotes punctuation that would otherwise be syntax', () => { - expect(quoteFtsTerm('cli.mjs')).toBe('"cli.mjs"') - expect(quoteFtsTerm('C++')).toBe('"C++"') - expect(quoteFtsTerm('say "hi"')).toBe('"say ""hi"""') - }) - - it('builds one phrase, an AND chain, and an OR chain from the same terms', () => { - expect(phraseExpression(['alpha', 'beta'])).toBe('"alpha beta"') - expect(andExpression(['alpha', 'beta'])).toBe('"alpha" AND "beta"') - expect(orExpression(['alpha', 'beta'])).toBe('"alpha" OR "beta"') - }) -}) diff --git a/src/main/ai-vault-search/session-search-query-planner.ts b/src/main/ai-vault-search/session-search-query-planner.ts deleted file mode 100644 index 6c2c2f3b91c..00000000000 --- a/src/main/ai-vault-search/session-search-query-planner.ts +++ /dev/null @@ -1,140 +0,0 @@ -import type { SessionSearchScope } from './session-search-engine-types' -import { identifierShadowTerms } from './session-search-identifier-split' - -// Tokens exactly as the unicode61 tokenizer with `_ . - / +` tokenchars emits them. -const INDEX_TOKEN = /[\p{L}\p{N}\p{M}\p{Co}_./+-]+/gu -const STOP_WORDS = new Set( - ( - 'a an and are as at be but by for from how i if in into is it its of on or that the this to ' + - 'was were what when where which who why with you your we my me do does did not no can could ' + - 'should would about our us they them there their has have had been being so such then than ' + - "these those there's im ive dont" - ).split(' ') -) -const MAX_BODY_TERMS = 48 -const MAX_TERMS = 64 - -// A query that quotes something from a transcript: camelCase, SCREAMING_SNAKE, -// a dotted or snake_case name, a path, a filename, a PR number, a ticket, code -// punctuation, or an error word. -const LITERAL_SHAPE = - /[A-Za-z0-9_]*[a-z][A-Z][A-Za-z0-9_]*|\b[A-Z][A-Z0-9]{2,}(_[A-Z0-9]+)+\b|\b\w{2,}[._]\w{2,}\b|\b[\w.-]+\/[\w/.-]+\b|\b\w+\.(ts|tsx|js|jsx|py|rs|go|json|md|sh|yml|yaml|toml|c|cc|h|java|sql)\b|#\d{3,}|\b[A-Z]{2,6}-\d{2,}\b|[(){};=]|::|->|--\w|\b(Error|Exception|Traceback|error:|warning:)\b/ -const QUOTED = /"[^"]{3,}"|'[^']{3,}'/ - -export type SessionSearchQueryPlan = { - literal: boolean - /** - * The query had more terms than the planner will search. What is dropped is - * the tail, so a match that only the last term would have found is missed; - * the caller is told rather than handed a confident empty answer. - */ - truncated: boolean - /** Deduplicated index-faithful terms for the OR fallback, incl. identifier pieces. */ - terms: string[] - /** Query-order tokens minus stop words: the phrase / AND candidate. */ - body: string[] -} - -export function isLiteralQuery(query: string): boolean { - return QUOTED.test(query) || LITERAL_SHAPE.test(query) -} - -/** - * The tokenizer contract, unfolded: the same boundaries FTS5 draws for - * `unicode61 tokenchars '_.-/+'`. Pinned against real `fts5vocab` output in - * session-search-fts5-contract.test.ts, which is what makes it safe to plan a - * query without asking SQLite. - */ -export function indexTokens(query: string, limit = Number.POSITIVE_INFINITY): string[] { - const out: string[] = [] - for (const match of query.matchAll(INDEX_TOKEN)) { - const token = match[0] - // Separators alone (`--`, `...`) are a token to FTS5 but never a search term. - if (/[\p{L}\p{N}\p{Co}]/u.test(token)) { - out.push(token) - if (out.length >= limit) { - break - } - } - } - return out -} - -/** - * `literal` overrides the shape test. Typo repair re-plans the query it - * corrected, and a corrected spelling can look like ordinary prose even though - * what was typed was a literal: `parseJsonn(the, data)` has the punctuation that - * makes it literal, `parsejson the data` does not. Without the override the - * re-plan would drop `the` as a stop word, so the repaired query would search - * for less than the original asked for and `repairedTerms` would report a body - * the user never typed. - */ -export function planSessionSearchQuery( - query: string, - literal = isLiteralQuery(query) -): SessionSearchQueryPlan { - // One past the cap, so the plan can tell a query that just fits from one that - // was cut. `indexTokens` stops at its limit, so it cannot be asked afterwards. - const overCap = indexTokens(query, MAX_BODY_TERMS + 1) - const truncated = overCap.length > MAX_BODY_TERMS - const raw = overCap.slice(0, MAX_BODY_TERMS) - let body = literal ? raw : raw.filter((token) => !STOP_WORDS.has(token.toLowerCase())) - if (body.length < 2) { - body = raw - } - const terms = [...new Set(body)] - const extra: string[] = [] - for (const term of terms) { - for (const piece of identifierShadowTerms(term, 12)) { - if (!terms.includes(piece) && !STOP_WORDS.has(piece) && !extra.includes(piece)) { - extra.push(piece) - } - } - } - return { - literal, - truncated, - terms: [...terms, ...extra].slice(0, MAX_TERMS), - body: body.slice(0, MAX_BODY_TERMS) - } -} - -// Why: `cli.mjs`, `foo-bar`, and `C++` are all FTS5 syntax errors unquoted. -export function quoteFtsTerm(term: string): string { - return `"${term.replaceAll('"', '""')}"` -} - -export function phraseExpression(terms: readonly string[]): string { - return quoteFtsTerm(terms.join(' ')) -} - -export function andExpression(terms: readonly string[]): string { - return terms.map(quoteFtsTerm).join(' AND ') -} - -export function orExpression(terms: readonly string[]): string { - return terms.map(quoteFtsTerm).join(' OR ') -} - -/** - * What a scope is, now that there is one FTS table. - * - * `conversation` used to be a second table holding a copy of the two prose - * columns. It is a column filter instead: PR 2 measured the filter at - * 1.16-1.36x the p95 of the dedicated table on a 105 MB corpus, against a 2x - * bar, and the table cost a tenth of the index to maintain. - * - * It lives beside the other expression builders, and not with the retrieval - * that uses it, because the typo repair has to ask the same question of the - * same scope and importing it from there is a cycle. - * - * The filter binds to the whole expression, so it is applied here and nowhere - * else — `{cols}: (a AND b)` filters both terms, while a prefix pasted in front - * of a bare `a AND b` would filter only `a` and quietly search tool output for - * the rest. - */ -const CONVERSATION_COLUMNS = '{user_text assistant_text}' - -export function scopedExpression(scope: SessionSearchScope, expression: string): string { - return scope === 'all' ? expression : `${CONVERSATION_COLUMNS}: (${expression})` -} diff --git a/src/main/ai-vault-search/session-search-query-schema.ts b/src/main/ai-vault-search/session-search-query-schema.ts deleted file mode 100644 index f17b290e530..00000000000 --- a/src/main/ai-vault-search/session-search-query-schema.ts +++ /dev/null @@ -1,79 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' -import { - SESSION_SEARCH_GENERATION_SQL, - SESSION_SEARCH_GENERATION_TRIGGERS -} from './session-search-index-generation' - -/** An engine feature the index on disk cannot serve. */ -export type SessionSearchUnavailableFeature = 'typo-repair' - -const QUERY_SCHEMA_SQL = ` --- The typo repair's whole dictionary. Why the index's own vocabulary and not a --- word list: it can never suggest a term this index does not hold, and it needs --- no model. fts5vocab is a view over the FTS5 b-tree, so it costs no extra rows. -CREATE VIRTUAL TABLE IF NOT EXISTS messages_vocab USING fts5vocab(messages_fts, 'row'); --- Locally logged queries, stored as typed, bounded. Nothing writes here unless a --- caller opts in; the eval set is rebuilt from it (see session-search-query-log). -CREATE TABLE IF NOT EXISTS search_log( - id INTEGER PRIMARY KEY, - ts TEXT NOT NULL, - query TEXT NOT NULL, - route TEXT NOT NULL, - hits INTEGER NOT NULL, - duration_ms REAL NOT NULL -); -${SESSION_SEARCH_GENERATION_SQL}` - -/** Everything the SQL above creates, so a missing one is what triggers a re-run. */ -const OWNED = ['messages_vocab', 'search_log', ...SESSION_SEARCH_GENERATION_TRIGGERS] - -/** - * The vocabulary's target. Creating a fts5vocab table over a missing FTS table - * succeeds and every query against it then fails, so the feature's health is - * this name's presence rather than the vocabulary's own. - */ -const VOCABULARY_SOURCE = 'messages_fts' - -const PROBED = [...OWNED, VOCABULARY_SOURCE] - -/** - * Creates whatever of the engine's own schema is missing, and reports what it - * still cannot serve. - * - * These objects are the query engine's, not the store's. Nothing on the write - * path reads any of them, so under the stack's YAGNI rule they do not belong in - * PR 2's schema, and an index built by a process that never opens an engine - * carries none of their cost. None of them needs a schema version either: every - * one is derived from what PR 2 already holds, so re-creating them over any of - * its files is correct, while a version bump would throw a whole index away to - * add a view over its own b-tree. - * - * Run per search, not once per engine. A capability is a fact about the file - * rather than about this object: another handle can rebuild the index under a - * live connection, so a verdict taken in a constructor is wrong for the rest of - * the engine's life in both directions — it would keep reaching for a table - * that went away and never pick one back up when it returned. The steady-state - * cost is the single indexed `sqlite_master` lookup below. - * - * A create that throws is not caught. The only way to reach one is an index - * whose `files` table is gone, which is a rebuild in flight — and an engine - * over that cannot report a hit's source either, so there is nothing to degrade - * to. Losing only the vocabulary's source is the case worth surviving, and that - * one is reported rather than thrown. - */ -export function ensureSessionSearchQuerySchema( - db: SyncDatabase -): readonly SessionSearchUnavailableFeature[] { - const present = presentNames(db) - if (OWNED.some((name) => !present.has(name))) { - db.exec(QUERY_SCHEMA_SQL) - } - return present.has(VOCABULARY_SOURCE) ? [] : ['typo-repair'] -} - -function presentNames(db: SyncDatabase): Set { - const rows = db - .prepare(`SELECT name FROM sqlite_master WHERE name IN (${PROBED.map(() => '?').join(',')})`) - .all(...PROBED) as { name: string }[] - return new Set(rows.map((row) => row.name)) -} diff --git a/src/main/ai-vault-search/session-search-retrieval.ts b/src/main/ai-vault-search/session-search-retrieval.ts deleted file mode 100644 index 2bbac5d3e4e..00000000000 --- a/src/main/ai-vault-search/session-search-retrieval.ts +++ /dev/null @@ -1,251 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' -import type { SessionSearchRoute, SessionSearchScope } from './session-search-engine-types' -import type { MessageRow, SessionRow } from './session-search-hit-ranking' -import { - andExpression, - orExpression, - phraseExpression, - planSessionSearchQuery, - scopedExpression, - type SessionSearchQueryPlan -} from './session-search-query-planner' -import type { SessionRowFilter } from './session-search-row-filter' -import { SessionSearchTypoRepair } from './session-search-typo-repair' - -// The operator-only walk: rows per page, and how far past a full candidate set -// it will read before giving up on finding more matches. -const RECENT_PAGE_ROWS = 512 -// Ids per `loadSessions` statement, with room to spare for the filter's own -// bound values beside them. -const SESSION_ID_BATCH = 500 -const RECENT_SCAN_FACTOR = 20 - -// Measured: user 3 / assistant 2 / tool 1 / identifiers 1 (MRR 0.503 vs 0.475 flat). -const FULL_WEIGHTS = '3.0, 2.0, 1.0, 1.0' -// The conversation scope zeroes the two columns its filter already excludes. -// Measured, and stated because it is easy to over-read: these zeros change no -// score. FTS5's bm25 sums over the columns the query matched, and the filter -// has already kept the match out of those two, so the same rows come back with -// `1.0, 1.0` here. They are a statement of what the scope means, not the fence -// that enforces it — `scopedExpression` is the fence. -const CONVERSATION_WEIGHTS = '3.0, 2.0, 0.0, 0.0' - -export type RetrievalScope = { - scope: SessionSearchScope - sort: 'relevance' | 'newest' - filter: SessionRowFilter - /** - * `repo:` / `path:`, which SQL cannot express. Applied over retrieved rows; - * see session-search-row-filter for why it cannot be pushed down. - */ - matchesOperators: (session: SessionRow) => boolean - /** - * Sessions retrieved before ranking cuts the page. See - * docs/reference/agent-session-search-query-tuning.md for the measurements - * behind the default; it is an option because the right value depends on how - * large an index is and no single number is right for every host. - */ - candidateLimit: number -} - -export type Retrieved = { - rows: MessageRow[] - route: SessionSearchRoute - /** The plan the rows were actually retrieved by; snippets highlight from it. */ - plan: SessionSearchQueryPlan - repairedTerms?: string[] -} - -/** - * The bm25 weights a scope ranks with. The conversation pair stays here rather - * than beside `scopedExpression`, because weights are a property of this SQL - * and nothing else asks for them. - */ -export function scopedWeights(scope: SessionSearchScope): string { - return scope === 'all' ? FULL_WEIGHTS : CONVERSATION_WEIGHTS -} - -/** The FTS half of a search: the route ladder and the SQL each rung runs. */ -export class SessionSearchRetrieval { - /** Null when this index has no vocabulary to repair against; the rung is skipped. */ - private readonly typoRepair: SessionSearchTypoRepair | null - - constructor( - private readonly db: SyncDatabase, - canRepairTypos = true - ) { - this.typoRepair = canRepairTypos ? new SessionSearchTypoRepair(db) : null - } - - /** - * The route ladder: phrase, then AND for a literal-looking query, then typo - * repair, then OR. - * - * Repair runs before the OR fallback rather than after it fails. A typo next - * to a common word would otherwise be masked: the common word alone retrieves - * plenty of rows over OR, so nothing would ever look like a miss worth - * repairing. - */ - run(plan: SessionSearchQueryPlan, scope: RetrievalScope): Retrieved { - const exact = this.literal(plan, scope) - if (exact) { - return { ...exact, plan } - } - const repaired = this.repair(plan, scope.scope) - const effective = repaired ?? plan - const literal = repaired ? this.literal(repaired, scope) : null - const found = literal ?? { - rows: this.match(orExpression(effective.terms), scope), - route: 'or' as const - } - return { - rows: found.rows, - route: repaired ? (`typo+${found.route}` as SessionSearchRoute) : found.route, - plan: effective, - ...(repaired ? { repairedTerms: repaired.body } : {}) - } - } - - /** - * Newest sessions the constraints allow: what an operator-only query names. - * - * Walked in pages rather than taken in one `LIMIT`, because the operators are - * applied in JS. A single cut of the newest N would hand ranking whatever - * happened to be recent and then throw most of it away, so `repo:x` on a busy - * index could answer with nothing while plenty matched. The walk is bounded - * both ways: it stops at a full candidate set, and at a ceiling on rows read. - */ - recent(scope: RetrievalScope): { sessions: SessionRow[]; incomplete: boolean } { - const { conditions, values } = scope.filter - const where = conditions.length > 0 ? `WHERE ${conditions.join(' AND ')}` : '' - const page = this.db.prepare( - `SELECT * FROM sessions ${where} - ORDER BY updated_at DESC, id DESC LIMIT ? OFFSET ?` - ) - const ceiling = scope.candidateLimit * RECENT_SCAN_FACTOR - const sessions: SessionRow[] = [] - let scanned = 0 - // Why the flag and not a count: both caps mean the same thing to a caller — - // a session it never saw may have matched — and only the loop knows which - // of them ended it. Reporting rows read instead let the engine infer - // completeness from a full candidate set alone, so giving up at the ceiling - // with nothing found looked exactly like a search that found nothing. - let incomplete = false - while (sessions.length < scope.candidateLimit) { - if (scanned >= ceiling) { - incomplete = true - break - } - const rows = page.all(...values, RECENT_PAGE_ROWS, scanned) as SessionRow[] - if (rows.length === 0) { - break - } - scanned += rows.length - for (const row of rows) { - if (sessions.length < scope.candidateLimit && scope.matchesOperators(row)) { - sessions.push(row) - } - } - } - return { sessions, incomplete: incomplete || sessions.length >= scope.candidateLimit } - } - - /** - * Read in batches, because the id list is as long as the candidate limit and - * every id is a bound parameter, so a single statement scales with a knob the - * tuning doc invites a host to raise. - * - * Not a fix for a reachable failure, and worth saying so: SQLite has bound - * `SQLITE_MAX_VARIABLE_NUMBER` at 32,766 since 3.32, every runtime this stack - * supports is past that, and the measured limit on this one is higher still. - * A candidate limit that large is not a configuration anyone would choose. - * The batch is here so the ceiling belongs to this file rather than to - * whichever SQLite the process happened to link. - */ - loadSessions(ids: readonly number[], scope: RetrievalScope): SessionRow[] { - const rows: SessionRow[] = [] - for (let start = 0; start < ids.length; start += SESSION_ID_BATCH) { - const batch = ids.slice(start, start + SESSION_ID_BATCH) - const conditions = [`id IN (${batch.map(() => '?').join(',')})`, ...scope.filter.conditions] - rows.push( - ...(this.db - .prepare(`SELECT * FROM sessions WHERE ${conditions.join(' AND ')}`) - .all(...batch, ...scope.filter.values) as SessionRow[]) - ) - } - return rows.filter((row) => scope.matchesOperators(row)) - } - - private repair( - plan: SessionSearchQueryPlan, - scope: SessionSearchScope - ): SessionSearchQueryPlan | null { - if (!this.typoRepair) { - return null - } - const typoRepair = this.typoRepair - let changed = false - const body = plan.body.map((term) => { - // Repaired inside the scope the search will run in, so a spelling only - // tool output carries neither suppresses a repair nor becomes one. - const fix = typoRepair.correct(term, scope) - if (fix && fix !== term.toLowerCase()) { - changed = true - return fix - } - return term - }) - // The repair changes spellings, not the query's character: the re-plan is - // told what the original decided so a corrected literal keeps every term it - // was typed with. - return changed ? planSessionSearchQuery(body.join(' '), plan.literal) : null - } - - /** Phrase, then AND, for literal-looking queries; null when neither matches. */ - private literal( - plan: SessionSearchQueryPlan, - scope: RetrievalScope - ): { rows: MessageRow[]; route: 'phrase' | 'and' } | null { - if (!plan.literal || plan.body.length === 0) { - return null - } - // A one-token literal (`resolveTerminalPath`, `src/a/b.ts`) is its own - // phrase: the tokenizer keeps it whole, so the exact token is the cheap, - // precise first try before the identifier pieces fan out over OR. - const phrase = this.match(phraseExpression(plan.body), scope) - if (phrase.length > 0) { - return { rows: phrase, route: 'phrase' } - } - if (plan.body.length < 2) { - return null - } - const and = this.match(andExpression(plan.body), scope) - return and.length > 0 ? { rows: and, route: 'and' } : null - } - - private match(expression: string, scope: RetrievalScope): MessageRow[] { - const { filter, sort, candidateLimit } = scope - const eligible = filter.conditions.length - ? ` AND m.session_row_id IN (SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')})` - : '' - const matched = `SELECT messages_fts.rowid AS rowid, - -bm25(messages_fts, ${scopedWeights(scope.scope)}) AS score, - m.session_row_id, m.role, m.ts, s.updated_at - FROM messages_fts JOIN messages m ON m.id = messages_fts.rowid - JOIN sessions s ON s.id = m.session_row_id WHERE messages_fts MATCH ?${eligible}` - // Why: collapse to one row per session BEFORE the candidate limit, on both - // sort orders, so a single long session cannot occupy the whole page. - // `max(score)` makes SQLite pick that session's best row for the bare columns. - // Cost of grouping instead of a bounded top-N sorter, measured: ~1.75x - // (49.6 vs 28.6 ms at 80k matching rows, 183.6 vs 104.1 ms at 240k) and a - // temp b-tree over every match. No inner LIMIT can bound it: the CTE has no - // order, so any cut drops whole sessions rather than their surplus rows. - const order = sort === 'newest' ? 'updated_at DESC, score DESC' : 'score DESC' - const sql = `WITH matched AS MATERIALIZED (${matched}) - SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched - GROUP BY session_row_id ORDER BY ${order} LIMIT ${candidateLimit}` - return this.db - .prepare(sql) - .all(scopedExpression(scope.scope, expression), ...filter.values) as MessageRow[] - } -} diff --git a/src/main/ai-vault-search/session-search-row-filter.test.ts b/src/main/ai-vault-search/session-search-row-filter.test.ts deleted file mode 100644 index dd7ca0d1e5c..00000000000 --- a/src/main/ai-vault-search/session-search-row-filter.test.ts +++ /dev/null @@ -1,137 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import type SyncDatabase from '../sqlite/sync-database' -import type { SessionSearchFilters } from './session-search-engine-types' -import { cwdKey } from './session-search-file-records' -import { sessionRowFilter } from './session-search-row-filter' -import { - openSessionSearchIndexFile, - type SessionSearchIndexFile -} from './session-search-index-test-fixture' - -let index: SessionSearchIndexFile | null = null - -afterEach(async () => { - await index?.close() - index = null -}) - -async function openIndex(): Promise { - index = await openSessionSearchIndexFile('ss-row-filter') - return index.db -} - -function addSession( - db: SyncDatabase, - id: number, - cwd: string | null, - overrides: { agent?: string; updatedAt?: string } = {} -): void { - db.prepare( - `INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,updated_at,resume_command) - VALUES (?,?,?,?,'fixture',?,?,?,'')` - ).run( - id, - overrides.agent ?? 'claude', - String(id), - `/synthetic/${id}`, - cwd, - cwdKey(cwd), - overrides.updatedAt ?? '2026-09-01T00:00:00.000Z' - ) -} - -function selected(db: SyncDatabase, filters: SessionSearchFilters = {}): number[] { - const filter = sessionRowFilter(filters) - const where = filter.conditions.length > 0 ? `WHERE ${filter.conditions.join(' AND ')}` : '' - return ( - db.prepare(`SELECT id FROM sessions ${where} ORDER BY id`).all(...filter.values) as { - id: number - }[] - ).map((row) => row.id) -} - -describe('a cwd scope is the sidebar key, or anything below it', () => { - it.each([ - ['C:\\Work\\App', 'c:/work/app', true], - ['C:\\Work\\App\\src', 'c:/work/app', true], - ['/work/APP/src', '/work/app', false], - ['/work/caf\u00e9', '/work/cafe\u0301', true], - ['/work/app-other', '/work/app', false], - ['/work/a_b/src', '/work/a_b', true], - ['/work/axb/src', '/work/a_b', false], - // Roots: `/` is the one key that is already a separator, which is where a - // range bound is easiest to get wrong. A Windows key is not under POSIX `/`. - ['/', '/', true], - ['/work/app', '/', true], - ['C:\\Work\\App', '/', false], - ['C:\\', 'C:\\', true], - ['C:\\Work\\App', 'C:\\', true] - ])('scopes %s under %s: %s', async (cwd, scope, expected) => { - const db = await openIndex() - addSession(db, 1, cwd) - expect(selected(db, { scopePaths: [scope] })).toEqual(expected ? [1] : []) - }) - - it('never matches a session whose transcript recorded no cwd', async () => { - const db = await openIndex() - addSession(db, 1, null) - expect(selected(db, { scopePaths: ['/work'] })).toEqual([]) - expect(selected(db)).toEqual([1]) - }) - - it('keeps a WSL UNC workspace distinct from the bare Linux spelling', async () => { - // PR 2 decided cwd_key does not qualify a Linux path with its distro: the - // collision is real but every SSH host has it too, and the fix is a column - // naming the execution host, not a key only some hosts spell differently. - const db = await openIndex() - addSession(db, 1, '\\\\wsl.localhost\\Ubuntu\\home\\ada\\app') - addSession(db, 2, '/home/ada/app') - expect(selected(db, { scopePaths: ['\\\\wsl$\\Ubuntu\\home\\ada'] })).toEqual([1]) - expect(selected(db, { scopePaths: ['/home/ada/app'] })).toEqual([2]) - expect(selected(db, { scopePaths: ['\\\\wsl$\\Debian\\home\\ada\\app'] })).toEqual([]) - }) -}) - -describe('caller filters', () => { - it('narrows by agent, and by updated-at floor', async () => { - const db = await openIndex() - addSession(db, 1, '/work/app', { agent: 'claude', updatedAt: '2026-09-01T00:00:00.000Z' }) - addSession(db, 2, '/work/app', { agent: 'codex', updatedAt: '2026-09-05T00:00:00.000Z' }) - expect(selected(db, { agents: ['codex'] })).toEqual([2]) - expect(selected(db, { since: '2026-09-03T00:00:00.000Z' })).toEqual([2]) - expect(selected(db, { agents: ['claude'], since: '2026-09-03T00:00:00.000Z' })).toEqual([]) - }) - - it('applies the retention cutoff through the files table', async () => { - const db = await openIndex() - addSession(db, 1, '/work/app') - addSession(db, 2, '/work/app') - db.prepare( - "INSERT INTO files(path,byte_offset,mtime_ms,session_row_id) VALUES ('a',0,100,1)" - ).run() - db.prepare( - "INSERT INTO files(path,byte_offset,mtime_ms,session_row_id) VALUES ('b',0,500,2)" - ).run() - const filter = sessionRowFilter({}, 300) - const rows = db - .prepare(`SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')}`) - .all(...filter.values) as { id: number }[] - expect(rows.map((row) => row.id)).toEqual([2]) - }) -}) - -it('plans a cwd scope as a seek on sessions_cwd_key, never a scan', async () => { - const db = await openIndex() - const filter = sessionRowFilter({ scopePaths: ['/work/app'] }) - const plan = ( - db - .prepare( - `EXPLAIN QUERY PLAN SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')}` - ) - .all(...filter.values) as { detail: string }[] - ).map((row) => row.detail) - - expect(plan.join(' | ')).toContain('sessions_cwd_key') - expect(plan.some((detail) => detail.startsWith('SEARCH'))).toBe(true) - expect(plan.some((detail) => detail.startsWith('SCAN sessions'))).toBe(false) -}) diff --git a/src/main/ai-vault-search/session-search-row-filter.ts b/src/main/ai-vault-search/session-search-row-filter.ts deleted file mode 100644 index 5015fa460e6..00000000000 --- a/src/main/ai-vault-search/session-search-row-filter.ts +++ /dev/null @@ -1,90 +0,0 @@ -import { cwdKey } from './session-search-file-records' -import type { SessionSearchFilters } from './session-search-engine-types' - -/** SQL fragments for the `sessions` WHERE clause; every condition is ANDed. */ -export type SessionRowFilter = { - conditions: string[] - values: (string | number)[] -} - -// Stored identity: `cwdKey` is the sidebar's `folderGroupKey` without its prefix, -// so a scope term and an indexed session are keyed by one function, never two. -const CWD = 'cwd_key' - -/** - * The narrowings SQL can express exactly, in one place, so retrieval, the - * operator-only page and the session load cannot drift apart. These conditions - * run over `sessions` itself. Reachability is not here and is not a condition: - * it is the INNER JOIN to `sessions` that every retrieval carries, which is - * what makes a message row a purge has not reclaimed yet unreadable. - * - * `repo:` and `path:` are deliberately absent. What they mean is the predicate - * the sessions panel applies (`matchesAiVaultQueryOperators`), and SQL cannot - * express it: LIKE folds ASCII and nothing else, so `path:CAFÉ` would miss - * `café`; `path:` searches the transcript path as well as the working - * directory, so `path:jsonl` would miss every session; and `repo:` compares the - * last two path segments, not one. A second spelling that came close would be a - * query meaning different things in the list and in the index, so the engine - * applies the panel's own predicate over the rows it retrieves instead. - * - * `scopePaths` stays here because it is exact: a prefix range over the key - * `cwdKey` produces, which folds exactly where the execution host folds — - * Windows drives, never a POSIX directory name. - */ -export function sessionRowFilter( - filters: SessionSearchFilters, - cutoffMs: number | null = null -): SessionRowFilter { - const filter: SessionRowFilter = { conditions: [], values: [] } - if (cutoffMs !== null) { - filter.conditions.push('id IN (SELECT session_row_id FROM files WHERE mtime_ms >= ?)') - filter.values.push(cutoffMs) - } - if (filters.agents && filters.agents.length > 0) { - filter.conditions.push(`agent IN (${filters.agents.map(() => '?').join(',')})`) - filter.values.push(...filters.agents) - } - if (filters.since) { - filter.conditions.push('updated_at >= ?') - filter.values.push(filters.since) - } - if (filters.scopePaths && filters.scopePaths.length > 0) { - // Several scopes mean any of them; every other narrowing is ANDed on. - const present = filters.scopePaths - .map((scope) => scopeCondition(filter, scope)) - .filter((condition) => condition !== null) - if (present.length > 0) { - filter.conditions.push(`(${present.join(' OR ')})`) - } - } - return filter -} - -/** A scope the caller could not key is a scope nothing is inside of. */ -function scopeCondition(filter: SessionRowFilter, scope: string): string | null { - const key = cwdKey(scope) - return key === null ? null : insideCondition(filter, key) -} - -/** - * `key` itself, or anything below it. Why a half-open range and not - * `substr(key, 1, length(?)) = ?`: only `>=`/`<` can seek `sessions_cwd_key`; - * the substr form scans it. The bound is the child prefix with its last byte - * incremented, so it stops at the end of that prefix and nowhere else. The two - * arms cannot merge: one range over the bare key would also swallow a sibling - * like `/work/app-other`. No wildcards, so `%`/`_` in a folder name are literal. - * - * The filesystem root is the one key that already ends in a separator, and - * appending a second one would bound the range at `//`, which sorts below every - * real child; `cwdKey` keeps it as `/` for exactly this reason. - */ -function insideCondition(filter: SessionRowFilter, key: string): string { - const children = key.endsWith('/') ? key : `${key}/` - filter.values.push(key, children, nextAfterPrefix(children)) - return `(${CWD} = ? OR (${CWD} >= ? AND ${CWD} < ?))` -} - -/** The first string that sorts after every string starting with `prefix`. */ -function nextAfterPrefix(prefix: string): string { - return prefix.slice(0, -1) + String.fromCharCode(prefix.charCodeAt(prefix.length - 1) + 1) -} diff --git a/src/main/ai-vault-search/session-search-sidebar-parity.test.ts b/src/main/ai-vault-search/session-search-sidebar-parity.test.ts deleted file mode 100644 index 081b16479ed..00000000000 --- a/src/main/ai-vault-search/session-search-sidebar-parity.test.ts +++ /dev/null @@ -1,137 +0,0 @@ -import { afterEach, expect, it } from 'vitest' -import type { AiVaultSession } from '../../shared/ai-vault-types' -import { filterAiVaultSessions } from '../../shared/ai-vault-session-filters' -import { - addSyntheticSession, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' - -// `repo:` and `path:` have to mean one thing. The sessions panel and the index -// answer from different stores by different mechanisms, so the only way to keep -// them equal is for both to run the same predicate; this asserts they do, over -// the shapes where a second SQL spelling went wrong. - -let harness: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - harness = null -}) - -type Fixture = { id: number; cwd: string; filePath: string; text: string } - -const SESSIONS: Fixture[] = [ - { - id: 1, - cwd: '/Users/Ada/orca/session-search', - filePath: '/Users/Ada/.claude/projects/a/one.jsonl', - text: 'harbor pilot manifest' - }, - { - id: 2, - cwd: '/Users/ada/work/café', - filePath: '/Users/ada/.codex/sessions/two.jsonl', - text: 'harbor dock crane' - }, - { - id: 3, - cwd: '/srv/other/service', - filePath: '/srv/.claude/projects/b/three.jsonl', - text: 'harbor manifest beta' - }, - { - id: 4, - cwd: 'C:\\Work\\Orca\\App', - filePath: 'C:\\Users\\Ada\\.claude\\four.jsonl', - text: 'harbor windows lane' - } -] - -// Each of these matched in the panel and missed in the index while the engine -// tried to say `repo:` / `path:` in SQL. -const QUERIES = [ - 'harbor path:jsonl', - 'harbor repo:orca/session-search', - 'harbor path:CAFÉ', - 'harbor path:/Users/Ada/orca', - 'harbor repo:app', - 'harbor repo:Orca/App', - 'harbor path:.codex', - 'harbor path:/srv repo:other/service', - 'harbor repo:session-search path:jsonl', - 'harbor path:"/Users/ada/work"', - 'harbor repo:nothing-here', - 'harbor path:one.jsonl path:two.jsonl', - 'harbor' -] - -function asSession(fixture: Fixture): AiVaultSession { - const at = '2026-09-01T00:00:00.000Z' - return { - id: String(fixture.id), - executionHostId: 'local', - agent: 'claude', - sessionId: String(fixture.id), - title: 'fixture', - cwd: fixture.cwd, - branch: null, - model: null, - filePath: fixture.filePath, - codexHome: null, - createdAt: at, - updatedAt: at, - modifiedAt: at, - messageCount: 1, - totalTokens: 0, - previewMessages: [{ role: 'user', text: fixture.text }], - queuedMessageCount: 0, - subagentTranscriptCount: 0, - resumeCommand: '', - subagent: null - } as AiVaultSession -} - -/** The panel's own answer, operators only: free text is FTS in the index. */ -function sidebarIds(query: string): string[] { - const operatorsOnly = query - .split(/\s+/) - .filter((token) => /^(repo|path):/i.test(token)) - .join(' ') - return filterAiVaultSessions(SESSIONS.map(asSession), { - query: operatorsOnly, - agents: ['claude'], - scope: 'all', - sort: 'updated', - activeWorktreePaths: [], - hideEmptySessions: false - }) - .map((session) => session.sessionId) - .sort() -} - -it.each(QUERIES)('answers %s the way the sessions panel does', async (query) => { - harness = await openSessionSearchHarness('ss-sidebar-parity') - for (const fixture of SESSIONS) { - addSyntheticSession(harness.db, { - id: fixture.id, - cwd: fixture.cwd, - text: fixture.text, - filePath: fixture.filePath, - sessionFilePath: fixture.filePath - }) - } - const engineIds = harness.engine - .search({ query, limit: 100 }) - .hits.map((hit) => hit.sessionId) - .sort() - expect(engineIds).toEqual(sidebarIds(query)) -}) - -it('is not vacuous: these queries do select, and reject, real sessions', () => { - // A parity suite where every query matched everything, or nothing, would pass - // against any predicate at all. - const answers = QUERIES.map((query) => sidebarIds(query).length) - expect(answers.some((count) => count > 0 && count < SESSIONS.length)).toBe(true) - expect(answers.some((count) => count === 0)).toBe(true) -}) diff --git a/src/main/ai-vault-search/session-search-snippet-marks.test.ts b/src/main/ai-vault-search/session-search-snippet-marks.test.ts deleted file mode 100644 index 71704b78fee..00000000000 --- a/src/main/ai-vault-search/session-search-snippet-marks.test.ts +++ /dev/null @@ -1,112 +0,0 @@ -import { afterEach, expect, it } from 'vitest' -import { - SESSION_SEARCH_SNIPPET_MARK_CLOSE, - SESSION_SEARCH_SNIPPET_MARK_OPEN -} from './session-search-engine-types' -import { - addSyntheticSession, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' - -// A snippet has to name which of a row's four columns matched, and the marks -// FTS5 wraps a match in are the only signal. Searching the marked text for the -// public `[[` reads a transcript's own brackets as a highlight — and transcripts -// are full of them, because a bash `[[ -f x ]]` and numpy's `[[1, 2]]` are -// exactly the sort of thing an agent session holds. Whether a column matched is -// the difference between two renderings of the same text instead. - -let harness: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - harness = null -}) - -const BASH = 'run this: if [[ -f /home/me/.aws/credentials ]]; then cat it; fi' -const TOOL = 'zebrafish appears only in the tool output here' - -it('shows the column that matched, not the one that happens to contain brackets', async () => { - harness = await openSessionSearchHarness('ss-snippet-marks') - // Session 1's match is in tool output while its user turn holds a bash test - // expression; session 2 is the same match with no brackets anywhere. - addSyntheticSession(harness.db, { id: 1, text: BASH, toolText: TOOL }) - addSyntheticSession(harness.db, { id: 2, text: 'run this script please', toolText: TOOL }) - - const hits = harness.engine.search({ query: 'zebrafish' }).hits - expect(hits).toHaveLength(2) - for (const hit of hits) { - expect(hit.evidence?.snippet).toContain( - `${SESSION_SEARCH_SNIPPET_MARK_OPEN}zebrafish${SESSION_SEARCH_SNIPPET_MARK_CLOSE}` - ) - expect(hit.evidence?.snippet).not.toContain('credentials') - } -}) - -it('falls back to any column for an identifier-only match, brackets or not', async () => { - // `zebra` reaches this row only through the identifier shadow column, which is - // what column -1 exists for. The user turn holds numpy output, so a bracket - // scan would have stopped at it and shown a column with no match in it. - harness = await openSessionSearchHarness('ss-snippet-marks-fallback') - addSyntheticSession(harness.db, { - id: 1, - text: 'numpy printed [[1, 2], [3, 4]] before the call', - toolText: 'zebra-fish-count = 4' - }) - - const [hit] = harness.engine.search({ query: 'zebra' }).hits - expect(hit?.evidence?.snippet).toContain( - `${SESSION_SEARCH_SNIPPET_MARK_OPEN}zebra${SESSION_SEARCH_SNIPPET_MARK_CLOSE}` - ) - expect(hit?.evidence?.snippet).not.toContain('numpy') -}) - -it('leaves a transcript’s own brackets in the text it shows', async () => { - // The marks are rewritten from private-use code points at the very end, so a - // row that both matches and contains `[[` keeps its own characters. - harness = await openSessionSearchHarness('ss-snippet-marks-literal') - addSyntheticSession(harness.db, { id: 1, text: `zebrafish ${BASH}` }) - - const [hit] = harness.engine.search({ query: 'zebrafish' }).hits - expect(hit?.evidence?.snippet).toContain( - `${SESSION_SEARCH_SNIPPET_MARK_OPEN}zebrafish${SESSION_SEARCH_SNIPPET_MARK_CLOSE}` - ) - expect(hit?.evidence?.snippet).toContain('[[ -f') -}) - -it('picks by comparison, so a private-use code point in content cannot pose as a mark', async () => { - // The marks are private-use code points, and a transcript may hold one: - // agent output carries Nerd Font glyphs, which live in the same block. So the - // column is chosen by comparing a marked rendering against an unmarked one, - // not by looking for a mark in the text. - harness = await openSessionSearchHarness('ss-snippet-marks-private-use') - addSyntheticSession(harness.db, { - id: 1, - text: 'the \uE000 glyph a font printed here', - toolText: TOOL - }) - - const [hit] = harness.engine.search({ query: 'zebrafish' }).hits - expect(hit?.evidence?.snippet).toContain('zebrafish') - expect(hit?.evidence?.snippet).not.toContain('glyph') -}) - -it('truncates on the last real mark, not on a bracket the transcript wrote', async () => { - // Over the character ceiling the snippet is cut, and it must not cut between - // an open mark and its close. Finding that open mark by searching for `[[` - // stops at the transcript's own bracket instead and throws away everything - // after it. - harness = await openSessionSearchHarness('ss-snippet-marks-truncation') - const long = (letter: string): string => - Array.from({ length: 5 }, () => `${letter.repeat(55)}/tail`).join(' ') - addSyntheticSession(harness.db, { - id: 1, - text: `zebrafish ${long('p')} [[ ${long('q')}` - }) - - const snippet = harness.engine.search({ query: 'zebrafish' }).hits[0]?.evidence?.snippet ?? '' - expect(snippet).toContain('[[zebrafish]]') - // The cut is the character ceiling, so the text after the transcript's own - // bracket survives up to it. - expect(snippet).toContain('qqqqq') -}) diff --git a/src/main/ai-vault-search/session-search-snippet.ts b/src/main/ai-vault-search/session-search-snippet.ts deleted file mode 100644 index 2e23e707dd0..00000000000 --- a/src/main/ai-vault-search/session-search-snippet.ts +++ /dev/null @@ -1,126 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' -import { - SESSION_SEARCH_SNIPPET_MARK_CLOSE, - SESSION_SEARCH_SNIPPET_MARK_OPEN -} from './session-search-engine-types' -import { - orExpression, - scopedExpression, - type SessionSearchQueryPlan -} from './session-search-query-planner' -import type { SessionSearchScope } from './session-search-engine-types' - -// What FTS5 wraps a match in before this module rewrites it to the public -// marks. Private-use code points, and not `[[`, because two different jobs here -// have to tell a mark from content: choosing the column to show, and refusing -// to cut a snippet between an open mark and its close. Transcripts contain -// `[[` — a bash `[[ -f x ]]`, numpy's `[[1, 2]]` — and a mark the content can -// forge makes both of those decisions wrong on real text. -const MARK_OPEN = '\uE000' -const MARK_CLOSE = '\uE001' - -const SNIPPET_TOKENS = 12 -// Why a ceiling on top of the token count: a transcript chunk can be 8000 -// characters with no separator in it, which FTS5 reports as one token, so -// "twelve tokens" is not by itself a bound on what a hit carries. -const SNIPPET_MAX_CHARS = 512 - -export type SessionSearchSnippet = { - text: string - truncated: boolean -} - -export const EMPTY_SNIPPET: SessionSearchSnippet = { text: '', truncated: false } - -/** - * The window of one message that shows why it matched. - * - * The expression is the plan's OR form rather than the route's, so a hit found - * through typo repair is marked with the repaired terms it was actually - * retrieved by, and a phrase hit still marks each of its words. - */ -export function sessionSearchSnippet( - db: SyncDatabase, - scope: SessionSearchScope, - rowid: number, - plan: SessionSearchQueryPlan -): SessionSearchSnippet { - // Why: the identifier shadow column is word soup; a hit that also matches in a - // prose column should be shown from there. Column -1 (any column) is the - // fallback for rows that only matched through the shadow column. - // - // The same four for every scope, because the scope is already in the - // expression below. A conversation snippet cannot come out of `tool_text` for - // the reason the search could not: the row has to match - // `{user_text assistant_text}: …` before any of these columns is read, and a - // row that matches under that filter carries its mark in column 0 or 1. A - // second list here would be a guard with nothing left to guard, and the two - // would mask each other's mistakes. - const columns = [0, 1, 2, -1] - // Each column twice: once marked, once with empty marks. Whether a column - // matched is then the difference between two renderings of the same text, - // which content cannot forge — searching the marked one for a mark reads a - // transcript's own `[[` as a highlight and shows a column that matched - // nothing. - const select = columns - .flatMap((column, index) => [ - `snippet(messages_fts, ${column}, '${MARK_OPEN}', '${MARK_CLOSE}', '…', ${SNIPPET_TOKENS}) AS c${index}`, - `snippet(messages_fts, ${column}, '', '', '…', ${SNIPPET_TOKENS}) AS p${index}` - ]) - .join(', ') - try { - // Why the subselect: a bound `rowid = ?` or `rowid IN (?)` next to MATCH is - // silently ignored by the FTS5 planner, which then returns the first match - // in the table. Why the join to `sessions`: retrieval proved this rowid - // belonged to a live session, but a purge can commit between that statement - // and this one, and a message row outlives its session row until the drain - // reaches it. INNER, never LEFT — this is the last read before content is - // returned to a caller. - const row = db - .prepare( - `SELECT ${select} FROM messages_fts - JOIN messages m ON m.id = messages_fts.rowid - JOIN sessions s ON s.id = m.session_row_id - WHERE messages_fts MATCH ? AND messages_fts.rowid IN (SELECT ?)` - ) - .get(scopedExpression(scope, orExpression(plan.terms)), rowid) as - | Record - | undefined - if (!row) { - return EMPTY_SNIPPET - } - // A snippet with nothing highlighted tells the user nothing; omit it. - const marked = columns - .map((_column, index) => row[`c${index}`]) - .find((text, index) => text !== undefined && text !== row[`p${index}`]) - return marked === undefined ? EMPTY_SNIPPET : publicMarks(truncateSnippet(marked)) - } catch { - return EMPTY_SNIPPET - } -} - -/** The internal marks, swapped for the ones a caller sees, once and at the end. */ -function publicMarks(snippet: SessionSearchSnippet): SessionSearchSnippet { - return { - ...snippet, - text: snippet.text - .replaceAll(MARK_OPEN, SESSION_SEARCH_SNIPPET_MARK_OPEN) - .replaceAll(MARK_CLOSE, SESSION_SEARCH_SNIPPET_MARK_CLOSE) - } -} - -/** Cut on a code-point boundary, and never between a mark and its close. */ -export function truncateSnippet(text: string): SessionSearchSnippet { - if (text.length <= SNIPPET_MAX_CHARS) { - return { text, truncated: false } - } - const points = [...text] - if (points.length <= SNIPPET_MAX_CHARS) { - return { text, truncated: false } - } - const cut = points.slice(0, SNIPPET_MAX_CHARS).join('') - const opened = cut.lastIndexOf(MARK_OPEN) - // An open mark with no close hands the renderer something it can never close. - const balanced = opened !== -1 && !cut.includes(MARK_CLOSE, opened) ? cut.slice(0, opened) : cut - return { text: balanced, truncated: true } -} diff --git a/src/main/ai-vault-search/session-search-source-presence.ts b/src/main/ai-vault-search/session-search-source-presence.ts deleted file mode 100644 index cc7155fccc8..00000000000 --- a/src/main/ai-vault-search/session-search-source-presence.ts +++ /dev/null @@ -1,40 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' -import type { SessionSearchSourcePresence } from './session-search-engine-types' - -/** - * Where each session's source stands, read from the index's own `files` table. - * - * Why not a stat: a search page of 20 hits would be 20 filesystem round trips - * on the query path, and on an SSH or WSL host each one can block for as long - * as the connection takes to answer — the reviewer's F11. The index already - * records what discovery last proved about every file it read, so the query - * path reads that instead of asking the disk again. - * - * The vocabulary is deliberately short of `missing`. A row here means the index - * holds a live file record for the session, which is `present`. No row means - * this read cannot tell whether the source is gone or merely unrecorded, and - * loss of contact is never evidence of absence - * (docs/reference/ssh-execution-boundary.md), so it is `unverifiable`. Proving - * a deletion is the indexer's job and it retires the session's rows outright. - */ -export function sessionSourcePresence( - db: SyncDatabase, - sessionRowIds: readonly number[] -): Map { - const presence = new Map( - sessionRowIds.map((id) => [id, 'unverifiable' as const]) - ) - if (sessionRowIds.length === 0) { - return presence - } - const rows = db - .prepare( - `SELECT DISTINCT session_row_id FROM files - WHERE session_row_id IN (${sessionRowIds.map(() => '?').join(',')})` - ) - .all(...sessionRowIds) as { session_row_id: number }[] - for (const row of rows) { - presence.set(row.session_row_id, 'present') - } - return presence -} diff --git a/src/main/ai-vault-search/session-search-typo-policy.test.ts b/src/main/ai-vault-search/session-search-typo-policy.test.ts deleted file mode 100644 index 938c1e1fc3e..00000000000 --- a/src/main/ai-vault-search/session-search-typo-policy.test.ts +++ /dev/null @@ -1,80 +0,0 @@ -import { describe, expect, it } from 'vitest' -import type SyncDatabase from '../sqlite/sync-database' -import { openSessionSearchIndexFile } from './session-search-index-test-fixture' -import { ensureSessionSearchQuerySchema } from './session-search-query-schema' -import { SessionSearchTypoRepair } from './session-search-typo-repair' - -/** A session row the planted messages below hang off, so a repair can see them. */ -function addSession(db: SyncDatabase, id: number): void { - db.prepare( - `INSERT INTO sessions(id,agent,session_id,file_path,title,resume_command) - VALUES (?, 'claude', ?, '/synthetic/fixture', 'typo fixture', '')` - ).run(id, String(id)) -} - -function addTerm(db: SyncDatabase, sessionRowId: number, term: string): void { - const rowid = db - .prepare("INSERT INTO messages(session_row_id, role) VALUES (?, 'user')") - .run(sessionRowId).lastInsertRowid - db.prepare('INSERT INTO messages_fts(rowid, user_text) VALUES (?, ?)').run(Number(rowid), term) -} - -describe('typo repair policy', () => { - it.each([ - { input: 'coalesces', candidate: 'coalesced', copies: 2, exact: true, expected: null }, - { input: 'coalescs', candidate: 'coalesces', copies: 1, exact: false, expected: null }, - { input: 'coalescs', candidate: 'coalesces', copies: 2, exact: false, expected: 'coalesces' }, - { input: 'café', candidate: 'cafe', copies: 1, exact: false, expected: null }, - { input: 'car', candidate: 'cars', copies: 2, exact: false, expected: null }, - { input: 'calm', candidate: 'clam', copies: 2, exact: false, expected: null } - ])( - 'repairs $input to $expected with $copies postings (exact=$exact)', - async ({ input, candidate, copies, exact, expected }) => { - const index = await openSessionSearchIndexFile('ss-typo-policy') - try { - ensureSessionSearchQuerySchema(index.db) - addSession(index.db, 1) - for (let i = 0; i < copies; i++) { - addTerm(index.db, 1, candidate) - } - if (exact) { - addTerm(index.db, 1, input) - } - expect(new SessionSearchTypoRepair(index.db).correct(input, 'all')).toBe(expected) - } finally { - await index.close() - } - } - ) - - // A purge cuts a session loose in one transaction and reclaims its rows over - // many, so the vocabulary can still list a term whose only rows nothing can - // reach. Abandoning the prefix at that term would lose a repair the rest of - // the index can already serve. - it('falls through to the best candidate a reader can still reach', async () => { - const index = await openSessionSearchIndexFile('ss-typo-orphaned') - try { - const { db } = index - ensureSessionSearchQuerySchema(db) - addSession(db, 1) - // `coalesces` scores higher against `coalescs` than `coalesced` does, and - // shares its prefix, so only the fall-through can reach the reachable one. - // Session 2 is never created: these rows are what an unfinished purge - // leaves behind, and the vocabulary counts them all the same. - for (const [term, session] of [ - ['coalesces', 2], - ['coalesces', 2], - ['coalesced', 1], - ['coalesced', 1] - ] as const) { - addTerm(db, session, term) - } - expect(db.prepare("SELECT doc FROM messages_vocab WHERE term='coalesces'").get()).toEqual({ - doc: 2 - }) - expect(new SessionSearchTypoRepair(db).correct('coalescs', 'all')).toBe('coalesced') - } finally { - await index.close() - } - }) -}) diff --git a/src/main/ai-vault-search/session-search-typo-repair.ts b/src/main/ai-vault-search/session-search-typo-repair.ts deleted file mode 100644 index 1aed90e2991..00000000000 --- a/src/main/ai-vault-search/session-search-typo-repair.ts +++ /dev/null @@ -1,163 +0,0 @@ -import type SyncDatabase from '../sqlite/sync-database' -import type { SessionSearchScope } from './session-search-engine-types' -import { quoteFtsTerm, scopedExpression } from './session-search-query-planner' - -// Why: a query term with zero postings is usually a typo. The index's own -// vocabulary (fts5vocab) is the dictionary, so repair needs no model and can -// never suggest a word the index does not contain. Measured MRR 0.553 → 0.566. -const MIN_TERM_LENGTH = 4 -const MAX_TERM_LENGTH = 40 -const LENGTH_SLACK = 2 -const MIN_DOC_FREQUENCY = 2 -const MIN_SIMILARITY = 0.82 -const MAX_CANDIDATES = 4000 -// Candidates counted against live rows per prefix before giving up on it. Only -// reached for a term the scope has no posting for, which is the rare case. -const MAX_VISIBILITY_PROBES = 8 -// How far a live count walks before it stops caring. It exists to break ties -// between candidates of equal similarity, and the difference between a term in -// sixty-four rows and one in six thousand does not change which is the better -// repair — but reading either in full would. -const MAX_COUNTED_ROWS = 64 - -// Longest common subsequence length; the indel distance is len(a)+len(b)-2·LCS. -function commonSubsequenceLength(a: string, b: string): number { - let previous = Array.from({ length: b.length + 1 }).fill(0) - let current = Array.from({ length: b.length + 1 }).fill(0) - for (let i = 1; i <= a.length; i += 1) { - for (let j = 1; j <= b.length; j += 1) { - current[j] = - a.charCodeAt(i - 1) === b.charCodeAt(j - 1) - ? previous[j - 1] + 1 - : Math.max(previous[j], current[j - 1]) - } - ;[previous, current] = [current, previous] - } - return previous[b.length] -} - -/** Normalized indel similarity in [0, 1], the scale rapidfuzz's `fuzz.ratio` uses. */ -function similarity(a: string, b: string): number { - const total = a.length + b.length - return total === 0 ? 1 : (2 * commonSubsequenceLength(a, b)) / total -} - -/** - * Spelling repair over the index's own vocabulary. - * - * The vocabulary proposes and a scoped count disposes. `messages_vocab` is a - * view over the whole FTS b-tree: it has no column filter, because fts5vocab is - * per table, and it counts rows whose session a purge already cut loose. So - * every decision that reaches the plan — whether a term is already spelled - * right, whether a candidate is eligible, and which of two equally close - * candidates wins — is taken from a `messages_fts MATCH` under the same column - * filter retrieval uses, joined to `sessions`. - * - * That is not tidiness. Reading the vocabulary directly made the repair depend - * on rows the search could never return: tool output suppressed a - * conversation-scope repair and supplied suggestions the scope would never - * show, and retention's orphan drain silently changed which word a query was - * repaired to. - * - * The cost is one bounded count per candidate examined, at most - * `MAX_VISIBILITY_PROBES` per prefix, and only for a term the scope has no - * posting for. See docs/reference/agent-session-search-query-tuning.md. - */ -export class SessionSearchTypoRepair { - private readonly liveRows: ReturnType - private readonly candidatesByPrefix: ReturnType - - constructor(db: SyncDatabase) { - this.liveRows = db.prepare( - `SELECT count(*) AS rows FROM ( - SELECT m.id FROM messages_fts - JOIN messages m ON m.id = messages_fts.rowid - JOIN sessions s ON s.id = m.session_row_id - WHERE messages_fts MATCH ? LIMIT ${MAX_COUNTED_ROWS})` - ) - // fts5vocab is ordered by term, so a prefix range plus a length band is a - // bounded scan and no sort. Ordered by term rather than by `doc`: the - // ordering decides which candidates survive the limit, and `doc` counts - // rows no reader can see, so the drain reclaiming them moved the cut. - this.candidatesByPrefix = db.prepare( - `SELECT term FROM messages_vocab - WHERE term >= ? AND term < ? AND length(term) BETWEEN ? AND ? - ORDER BY term LIMIT ?` - ) - } - - /** Live rows carrying this term inside `scope`, counted no further than it matters. */ - private countRows(term: string, scope: SessionSearchScope): number { - const row = this.liveRows.get(scopedExpression(scope, quoteFtsTerm(term))) as { rows: number } - return row.rows - } - - /** Whether a live row inside `scope` holds this term. */ - hasPostings(term: string, scope: SessionSearchScope): boolean { - return this.countRows(term, scope) > 0 - } - - /** Returns the closest indexed term, or null when `term` exists or nothing is close enough. */ - correct(term: string, scope: SessionSearchScope): string | null { - const lowered = term.toLowerCase() - if (lowered.length < MIN_TERM_LENGTH || lowered.length > MAX_TERM_LENGTH) { - return null - } - if (this.hasPostings(lowered, scope)) { - return null - } - // Two-letter prefix first (a typo rarely hits both), then the transposed - // pair, then the bare first letter as the wide fallback. - const prefixes = [lowered.slice(0, 2), lowered[1] + lowered[0], lowered[0]] - for (const prefix of prefixes) { - const best = this.bestVisible(lowered, prefix, scope) - if (best) { - return best - } - } - return null - } - - /** - * The closest candidate at `prefix` that this scope can actually answer with. - * - * Ranking is pure CPU, so the walk is bounded rather than the count: the - * closest term can be one the scope never shows, and abandoning the prefix - * there would lose a repair the rest of the index can serve. Ties on - * similarity go to the more common word, which is the same prior the - * vocabulary's `doc` used to supply — counted live here so the answer does - * not move when a purge reclaims rows nothing could reach. - */ - private bestVisible(lowered: string, prefix: string, scope: SessionSearchScope): string | null { - const counted = this.ranked(lowered, prefix) - .slice(0, MAX_VISIBILITY_PROBES) - .map((candidate) => ({ ...candidate, rows: this.countRows(candidate.term, scope) })) - .filter((candidate) => candidate.rows >= MIN_DOC_FREQUENCY) - if (counted.length === 0) { - return null - } - // Already sorted by similarity; a stable sort keeps that and orders the ties. - return counted.sort((left, right) => right.score - left.score || right.rows - left.rows)[0]! - .term - } - - /** Candidates similar enough to be a repair, closest first. */ - private ranked(lowered: string, prefix: string): { term: string; score: number }[] { - return this.candidates(prefix, lowered.length) - .map((row) => ({ term: row.term, score: similarity(lowered, row.term) })) - .filter((candidate) => candidate.score >= MIN_SIMILARITY) - .sort((left, right) => right.score - left.score || (left.term < right.term ? -1 : 1)) - } - - private candidates(prefix: string, length: number): { term: string }[] { - const last = prefix.charCodeAt(prefix.length - 1) - const upper = prefix.slice(0, -1) + String.fromCharCode(last + 1) - return this.candidatesByPrefix.all( - prefix, - upper, - Math.max(MIN_TERM_LENGTH - 1, length - LENGTH_SLACK), - length + LENGTH_SLACK, - MAX_CANDIDATES - ) as { term: string }[] - } -} diff --git a/src/main/ai-vault-search/session-search-typo-scope.test.ts b/src/main/ai-vault-search/session-search-typo-scope.test.ts deleted file mode 100644 index 98e3938178e..00000000000 --- a/src/main/ai-vault-search/session-search-typo-scope.test.ts +++ /dev/null @@ -1,71 +0,0 @@ -import { afterEach, expect, it } from 'vitest' -import { - addSyntheticSession, - openSessionSearchHarness, - type SessionSearchHarness -} from './session-search-engine-test-fixture' - -// Typo repair used to read `messages_vocab` and probe `messages_fts` with no -// column filter, so tool output decided whether a conversation-scoped query was -// repaired — in both directions. A tool row carrying the misspelling made the -// query look correctly spelled and suppressed the repair; a tool row carrying a -// rare word offered it as the suggestion, naming in `repairedTerms` a string -// from a column the scope will never show. - -let harness: SessionSearchHarness | null = null -let control: SessionSearchHarness | null = null - -afterEach(async () => { - await harness?.close() - await control?.close() - harness = null - control = null -}) - -it('repairs a conversation query the same way with or without a tool row', async () => { - harness = await openSessionSearchHarness('ss-typo-scope-suppress') - addSyntheticSession(harness.db, { id: 1, text: 'we changed resolveTerminalPath today', rows: 2 }) - // A second session whose tool output happens to contain the misspelling. - addSyntheticSession(harness.db, { - id: 2, - text: 'ran the linter', - toolText: 'warning: unknown symbol resolveterminalpth in build log', - rows: 2, - role: 'assistant' - }) - - // The same index without that one tool row. - control = await openSessionSearchHarness('ss-typo-scope-control') - addSyntheticSession(control.db, { id: 1, text: 'we changed resolveTerminalPath today', rows: 2 }) - addSyntheticSession(control.db, { id: 2, text: 'ran the linter' }) - - const request = { query: 'resolveterminalpth', scope: 'conversation' } as const - const withTool = harness.engine.search(request) - const clean = control.engine.search(request) - - expect(clean.planner.repairedTerms).toEqual(['resolveterminalpath']) - expect(clean.hits.map((hit) => hit.sessionId)).toEqual(['1']) - expect(withTool.planner.repairedTerms).toEqual(clean.planner.repairedTerms) - expect(withTool.hits.map((hit) => hit.sessionId)).toEqual(clean.hits.map((hit) => hit.sessionId)) -}) - -it('never repairs a conversation query onto a word only tool output holds', async () => { - harness = await openSessionSearchHarness('ss-typo-scope-leak') - addSyntheticSession(harness.db, { - id: 1, - text: 'ran the deploy', - toolText: 'AWS_SESSION_TOKEN=quicksilverfox expired', - rows: 2, - role: 'assistant' - }) - addSyntheticSession(harness.db, { id: 2, text: 'ordinary prose about nothing' }) - - const narrowed = harness.engine.search({ query: 'quicksilverfx', scope: 'conversation' }) - expect(narrowed.planner.repairedTerms).toBeUndefined() - expect(narrowed.hits).toEqual([]) - // The same query over the whole corpus still finds it, which is the scope - // doing its job rather than the repair being broken. - const wide = harness.engine.search({ query: 'quicksilverfx', scope: 'all' }) - expect(wide.planner.repairedTerms).toEqual(['quicksilverfox']) - expect(wide.hits.map((hit) => hit.sessionId)).toEqual(['1']) -}) diff --git a/src/shared/ai-vault-search-query-operators.test.ts b/src/shared/ai-vault-search-query-operators.test.ts deleted file mode 100644 index 0bb5cbc463b..00000000000 --- a/src/shared/ai-vault-search-query-operators.test.ts +++ /dev/null @@ -1,119 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { parseVaultQuery } from './ai-vault-session-filters' -import { - hasAiVaultSearchQueryOperators, - splitAiVaultSearchQuery -} from './ai-vault-search-query-operators' - -describe('what counts as an operator', () => { - it('splits repo: and path: out of the free text', () => { - const split = splitAiVaultSearchQuery('relay capacity repo:orca path:/work/app') - expect(split.text).toBe('relay capacity') - expect(split.terms).toEqual(['relay', 'capacity']) - expect(split.repoTerms).toEqual(['orca']) - expect(split.pathTerms).toEqual(['/work/app']) - expect(hasAiVaultSearchQueryOperators(split)).toBe(true) - }) - - it('keeps a value that only looks like an operator as ordinary text', () => { - const split = splitAiVaultSearchQuery('myrepo:x https://host/path:y') - expect(split.repoTerms).toEqual([]) - expect(split.pathTerms).toEqual([]) - expect(split.text).toBe('myrepo:x https://host/path:y') - }) - - it('reads a quoted operator value whole, including its spaces', () => { - expect(splitAiVaultSearchQuery('path:"/Users/ada/My Project" needle').pathTerms).toEqual([ - '/Users/ada/My Project' - ]) - }) - - it('does not let an apostrophe in prose swallow the operator between quotes', () => { - const split = splitAiVaultSearchQuery("it's a repo:orca thing's") - expect(split.repoTerms).toEqual(['orca']) - }) - - it('preserves operator case, which the panel folds and the index must not', () => { - // cwd_key keeps execution-host case, so folding here would lose a POSIX - // directory whose name differs only in case. - expect(splitAiVaultSearchQuery('path:/Work/App').pathTerms).toEqual(['/Work/App']) - expect(parseVaultQuery('path:/Work/App').pathTerms).toEqual(['/work/app']) - }) - - it('has no operators when the query is plain text', () => { - expect(hasAiVaultSearchQueryOperators(splitAiVaultSearchQuery('relay capacity'))).toBe(false) - }) -}) - -// The panel parses through this module now, so the two cannot disagree by -// construction. What is worth pinning is the handful of shapes where the -// panel's old hand-rolled tokenizer answered differently, so the change of -// behaviour is a decision on the record rather than a surprise. -describe('the shapes where the panel parser used to answer differently', () => { - it.each([ - ['repo:"" x', 'repoTerms'], - ['path:"" x', 'pathTerms'] - ] as const)('drops the empty operator value in %s instead of filtering on `""`', (query, key) => { - // The old tokenizer kept the quote characters as the value, so `repo:""` - // filtered on a label no session has and silently emptied the list. An - // operator with nothing in it is not a narrowing. - expect(splitAiVaultSearchQuery(query)[key]).toEqual([]) - expect(parseVaultQuery(query)[key]).toEqual([]) - }) - - it.each([ - ['repo:" " x', 'repoTerms'], - ['path:" " x', 'pathTerms'] - ] as const)('drops the whitespace-only operator value in %s too', (query, key) => { - // Same defect as `repo:""` wearing a different hat: an untrimmed `" "` - // survives as a term, matches no label, and empties the list. - expect(splitAiVaultSearchQuery(query)[key]).toEqual([]) - expect(parseVaultQuery(query)[key]).toEqual([]) - }) - - it('trims a quoted operator value rather than searching for the spaces', () => { - expect(splitAiVaultSearchQuery('repo:" session-search "').repoTerms).toEqual(['session-search']) - }) - - it.each(['"" empty', "'' empty", '" " empty'])( - 'reads the empty quotes in %s as an empty term', - (query) => { - // Same reason one level up: the old parser searched for the two characters - // and found nothing, where an empty term matches everything and leaves the - // rest of the query to do the work. - expect(parseVaultQuery(query).terms).toEqual(['', 'empty']) - } - ) - - it.each([ - ['"foo"bar', { terms: ['foo', 'bar'], repoTerms: [], pathTerms: [] }], - ['"a b"c', { terms: ['a b', 'c'], repoTerms: [], pathTerms: [] }], - ['repo:"a"b', { terms: ['b'], repoTerms: ['a'], pathTerms: [] }], - ['path:"a"b', { terms: ['b'], repoTerms: [], pathTerms: ['a'] }], - ['repo:"a b"c d', { terms: ['c', 'd'], repoTerms: ['a b'], pathTerms: [] }] - ])('reads %s exactly as the panel always has', (query, expected) => { - // A closing quote does not have to end a word. Requiring it turned each of - // these into one term carrying its own quote characters, which matches - // nothing; the apostrophe case below is protected by the token start, not - // by that rule. - expect(parseVaultQuery(query)).toEqual(expected) - }) -}) - -describe('agrees with the sessions panel parser on operator recognition', () => { - it.each([ - 'relay capacity', - 'repo:orca needle', - 'path:/work/app needle', - 'myrepo:x', - 'needle repo:orca path:/work/app', - 'path:"/Users/ada/My Project"', - 'https://host/path:y' - ])('reads the same operators out of %s', (query) => { - const split = splitAiVaultSearchQuery(query) - const parsed = parseVaultQuery(query) - const fold = (values: readonly string[]): string[] => values.map((v) => v.toLowerCase()).sort() - expect(fold(split.repoTerms)).toEqual(fold(parsed.repoTerms)) - expect(fold(split.pathTerms)).toEqual(fold(parsed.pathTerms)) - }) -}) diff --git a/src/shared/ai-vault-search-query-operators.ts b/src/shared/ai-vault-search-query-operators.ts deleted file mode 100644 index a75da8c769e..00000000000 --- a/src/shared/ai-vault-search-query-operators.ts +++ /dev/null @@ -1,90 +0,0 @@ -/** Anchored at a token start only, so `myrepo:x` and `https://h/path:x` stay literal. */ -const OPERATOR = /(repo|path):/iy - -export type AiVaultSearchQuerySplit = { - /** Query minus the operator tokens, quoting intact; what FTS sees. */ - text: string - /** The same free text as tokens with quotes stripped; what a substring matcher wants. */ - terms: readonly string[] - /** Operator values as typed apart from surrounding space: the panel folds case, the index does not. */ - repoTerms: readonly string[] - pathTerms: readonly string[] -} - -/** - * The one reading of `repo:` / `path:` in the product: the sessions panel and the - * search index must agree on what is an operator and what is ordinary text. - */ -export function splitAiVaultSearchQuery(query: string): AiVaultSearchQuerySplit { - const spans: string[] = [] - const terms: string[] = [] - const repoTerms: string[] = [] - const pathTerms: string[] = [] - let index = 0 - while (index < query.length) { - if (isBoundary(query[index])) { - index += 1 - continue - } - OPERATOR.lastIndex = index - const operator = OPERATOR.exec(query) - if (operator) { - const at = index + operator[0].length - const quoted = readQuoted(query, at) - const value = quoted?.value ?? readBare(query, at) - index = quoted ? quoted.end : at + value.length - // Trimmed for the same reason an empty value is dropped: `repo:" "` is - // not a narrowing anyone typed on purpose, and an untrimmed one matches - // no label at all, which silently empties the list. - const operand = value.trim() - if (operand) { - ;(operator[1]!.toLowerCase() === 'repo' ? repoTerms : pathTerms).push(operand) - } - continue - } - const quoted = readQuoted(query, index) - const value = quoted?.value ?? readBare(query, index) - const end = quoted ? quoted.end : index + value.length - spans.push(query.slice(index, end)) - // The span keeps the query verbatim for FTS; only the substring matcher's - // copy is trimmed, so `" "` reads as the empty term `""` already does - // rather than as a term no session's text contains. - terms.push(value.trim()) - index = end - } - return { text: spans.join(' '), terms, repoTerms, pathTerms } -} - -export function hasAiVaultSearchQueryOperators(split: AiVaultSearchQuerySplit): boolean { - return split.repoTerms.length > 0 || split.pathTerms.length > 0 -} - -function isBoundary(char: string | undefined): boolean { - return char === undefined || /\s/.test(char) -} - -/** - * A quoted span, or null when this is not one. - * - * What keeps the apostrophes in `it's a repo:orca thing's` from opening a span - * that swallows the operator is the caller: this only ever runs at a token - * start, and the quote in `it's` is not at one. The closing quote is then just - * the next one, wherever it falls, so `"a b"c` reads as the panel has always - * read it — the span, then the rest as its own token. - */ -function readQuoted(query: string, at: number): { value: string; end: number } | null { - const quote = query[at] - if (quote !== '"' && quote !== "'") { - return null - } - const close = query.indexOf(quote, at + 1) - return close === -1 ? null : { value: query.slice(at + 1, close), end: close + 1 } -} - -function readBare(query: string, at: number): string { - let end = at - while (end < query.length && !isBoundary(query[end])) { - end += 1 - } - return query.slice(at, end) -} diff --git a/src/shared/ai-vault-session-filters.ts b/src/shared/ai-vault-session-filters.ts index 39aedaf4626..7a0708151ed 100644 --- a/src/shared/ai-vault-session-filters.ts +++ b/src/shared/ai-vault-session-filters.ts @@ -8,7 +8,6 @@ import { normalizeRuntimePathSeparators } from './cross-platform-path' import { isClipboardTextByteLengthOverLimit } from './clipboard-text' -import { splitAiVaultSearchQuery } from './ai-vault-search-query-operators' import { parseWslUncPath } from './wsl-paths' import type { AiVaultAgent, @@ -180,61 +179,31 @@ export function agentLabel(agent: AiVaultAgent): string { return aiVaultAgentLabel(agent) } -/** - * One reading of `repo:` / `path:` for the whole product. - * - * Delegates to `splitAiVaultSearchQuery`, which the search index also plans - * from, so a query cannot mean one thing in this list and another in the index. - * The values come back folded because everything this file compares is folded; - * the index keeps the unfolded form, which is why the split itself does not. - */ export function parseVaultQuery(query: string): ParsedQuery { - const split = splitAiVaultSearchQuery(query) - const fold = (values: readonly string[]): string[] => values.map((value) => value.toLowerCase()) - return { - terms: fold(split.terms), - repoTerms: fold(split.repoTerms), - pathTerms: fold(split.pathTerms) - } -} + const terms: string[] = [] + const repoTerms: string[] = [] + const pathTerms: string[] = [] -/** What `repo:` and `path:` are compared against for one session. */ -export type AiVaultQueryOperatorTarget = { - cwd: string | null - filePath: string - /** - * What `repo:` matches. The panel passes a resolved project label when it has - * one; everything else falls back to the last two path segments. - */ - repoLabel?: string -} + for (const rawToken of tokenizeQuery(query)) { + const token = rawToken.toLowerCase() + if (token.startsWith('repo:')) { + const value = token.slice('repo:'.length) + if (value) { + repoTerms.push(value) + } + continue + } + if (token.startsWith('path:')) { + const value = token.slice('path:'.length) + if (value) { + pathTerms.push(value) + } + continue + } + terms.push(token) + } -/** - * Whether one session satisfies every `repo:` and `path:` term. - * - * The single definition of what those operators mean. The search index applies - * this over its retrieved rows rather than expressing it in SQL, because SQL - * cannot: LIKE folds ASCII and nothing else, and `path:` searches the transcript - * path as well as the working directory. Both keys are conjunctive, matching - * the qualifier semantics the panel has always had. - */ -export function matchesAiVaultQueryOperators( - target: AiVaultQueryOperatorTarget, - operators: { repoTerms: readonly string[]; pathTerms: readonly string[] } -): boolean { - if (operators.repoTerms.length > 0) { - const repoLabel = (target.repoLabel ?? folderLabel(target.cwd)).toLowerCase() - if (operators.repoTerms.some((term) => !repoLabel.includes(term.toLowerCase()))) { - return false - } - } - if (operators.pathTerms.length > 0) { - const pathSearch = `${target.cwd ?? ''} ${target.filePath}`.toLowerCase() - if (operators.pathTerms.some((term) => !pathSearch.includes(term.toLowerCase()))) { - return false - } - } - return true + return { terms, repoTerms, pathTerms } } function matchesQuery( @@ -260,18 +229,25 @@ function matchesQuery( return false } } - const sessionProject = filters.sessionProjectById?.get(session.id) - return matchesAiVaultQueryOperators( - { - cwd: session.cwd, - filePath: session.filePath, - repoLabel: - sessionProject?.kind === 'repo' - ? (filters.projectLabelByKey?.get(sessionProject.key) ?? sessionProject.label) - : undefined - }, - parsed - ) + if (parsed.repoTerms.length > 0) { + const sessionProject = filters.sessionProjectById?.get(session.id) + const repoLabel = ( + sessionProject?.kind === 'repo' + ? (filters.projectLabelByKey?.get(sessionProject.key) ?? sessionProject.label) + : folderLabel(session.cwd) + ).toLowerCase() + if (parsed.repoTerms.some((term) => !repoLabel.includes(term))) { + return false + } + } + if (parsed.pathTerms.length > 0) { + const pathSearch = `${session.cwd ?? ''} ${session.filePath}`.toLowerCase() + if (parsed.pathTerms.some((term) => !pathSearch.includes(term))) { + return false + } + } + + return true } function sessionSortTime(session: AiVaultSession, sort: AiVaultSort): number { @@ -315,3 +291,25 @@ function createAiVaultWorkspaceMatcher(workspacePath: string): (normalizedCwd: s const matchesLinux = createNormalizedPathInsideOrEqualMatcher(workspaceWslPath.linuxPath) return (cwd) => matches(cwd) || matchesLinux(cwd) } + +function tokenizeQuery(query: string): string[] { + const tokens: string[] = [] + // Why: keep quoted operator values (repo:/path:) intact so labels and paths + // containing spaces still match — e.g. path:"/Users/ada/My Project". + const pattern = /(repo|path):"([^"]+)"|(repo|path):'([^']+)'|"([^"]+)"|'([^']+)'|(\S+)/gi + let match: RegExpExecArray | null + while ((match = pattern.exec(query)) !== null) { + const operator = match[1] ?? match[3] + const operatorValue = match[2] ?? match[4] + if (operator && operatorValue?.trim()) { + tokens.push(`${operator.toLowerCase()}:${operatorValue.trim()}`) + continue + } + + const token = match[5] ?? match[6] ?? match[7] + if (token?.trim()) { + tokens.push(token.trim()) + } + } + return tokens +} From 1fedda14caa29bf88b609c2035fd1fc04d6daea6 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 11 Sep 2026 01:24:22 -0400 Subject: [PATCH 3/4] feat(ai-vault-search): session search indexer library owning backfill, reconcile and retention (#19751) * feat(ai-vault-search): expose the seams a scheduler needs to stay honest Three questions a lifecycle owner has to be able to ask, none of which had an answer: how many unfinished writes this open tombstoned, what the index believes it holds, and whether the session list's cursor already covers a file. The last one is not a nicety. The reader only opens a transcript the list still needs, so any file the list scanned before the index existed would be reused from cache and never reach the index at all. Naming the rule where it lives keeps one spelling of it instead of two. * feat(ai-vault-search): the bounds an unasked background index has to respect An injected clock, a retention window, a per-cycle allowance in files and bytes, a bounded re-read queue, and the load-aware pacing from the original branch. The allowance deliberately does not carry unspent room forward: accumulating it would let a long idle stretch buy one unbounded cycle, which is the stall the budget exists to prevent. Work that does not fit is queued, not dropped, and a transcript larger than a whole cycle's bytes is admitted alone rather than being refused forever. * feat(ai-vault-search): read a candidate set into the index, and say what happened One pass over a candidate list, plus the four things a caller has to be told about afterwards: what was discovered, which roots could not be read, which sources are provably gone, and where the whole run stands. Two absence rules are load-bearing. The file walker swallows a readdir failure and returns, so an unreadable root and an uninstalled agent both arrive as "no files"; only a root that yielded nothing is re-probed, and only ENOENT counts as absent. A source is retired on the same evidence and no weaker: an EACCES, an EIO or a stalled distro keeps its rows. Progress lives in the index's own files table, so a pass skips what it already covers and an interrupted run resumes instead of starting over. * feat(ai-vault-search): a whole-machine sweep and a recency-window cycle The sweep enumerates every root without a limit and is the only pass that can retire a source deleted while nothing was running. The cycle re-stats the newest N per agent, which is the sidebar's own rule rather than a second one, folds in the store's stale set and anything a caller invalidated, and reads what changed inside the cycle's allowance. A replaced file gets a whole re-read in the cycle that notices it. Waiting for the index to decline an append and mark itself stale would cost a second cycle and, because the reader resumes from the session list's cursor, would decline again every cycle after that. * feat(ai-vault-search): SessionSearchIndexer, with its freshness claim tested The object that owns freshness for the index: a full sweep on start, a timer that reconciles the recent window, retention that purges when it narrows and re-sweeps when it widens, and a pause that stops the writers rather than only the timer. A library, not a service. No Electron, no app lifecycle, no settings read, no IPC, and nothing constructs it. The clock is injected because a guarantee stated in wall time is a claim until a test can advance the clock and watch it hold: a transcript in the recent window that grows, is rename-replaced, or is deleted is reflected within one interval, each with its own test. * test(ai-vault-search): index a conversation held in Orca's own chat Reviewer F4 and the plan's fourth open decision. A conversation held in the panel writes the same file in the same place as one held in the terminal, so it must be searchable through the same path with nothing else running. The test constructs the indexer over an isolated root, writes and then appends native-chat-shaped rows, advances the clock one interval, and reads the rows back through the published views. * fix(ai-vault-search): ask the store before reading its cursor The pass read the index's own file row to decide how to read a candidate, and only then asked whether the store would accept it. A store that is paused or already closed answers no to everything, so those cursor reads were against a handle it had given up. * feat(ai-vault-search): take the reader's whole-read seam and PR 2's pause rules `requestWholeTranscriptRead` replaces the local invalidation: the reader owns the resume point, so asking it is the honest way to say the index needs a span the session list has already moved past. Two callers, and the second is the one no decline can reach. A forced path is one the store handed back from `takeStale` or a caller invalidated, and it has to arrive as a `replace` or the consumer declines the same append forever. But when the list's cursor already sits at a file's current stat the parse opens nothing at all, so no consumer is asked and there is nothing to record. That is every transcript on the machine the first time the index is switched on inside a running app, so it gets a test on both the sweep and the cycle path. Pausing now keeps the store's re-read set, so `filesPending` and `droppedPending` add both bounded queues together: a caller cannot act on one of them alone. The schema creates the index directory, so the indexer no longer does. * refactor(ai-vault-search): one ceiling for both re-read queues The indexer's scheduled-work queue carried a bound of its own beside the store's STALE_PATH_LIMIT, so `droppedPending` summed two numbers that meant two different things. It now shares the store's ceiling, set by the indexer, which is the only thing holding both queues. They stay separate queues. The store records that the index has a hole in a file; this records work scheduled and not yet done. Merging them by pushing budget leftovers through `markStale` would turn every deferred append into a whole re-read. * fix(ai-vault-search): stop the indexer from going quiet and calling it current Seven review findings, all in the seam between "stopped working" and "finished". One commit because they meet in the same three files. A pause part way through a backfill abandoned it. The flag was cleared on entry, the abort was swallowed as a normal stop, and resume only re-swept when the pause outlasted an interval. A sweep is now due until one completes, and work drained out of a queue and then not read goes back: a stale or invalidated path is a hole in the index, not finished work. The retention cutoff was set once at construction while purges took a fresh one, so with a one-day window a sweep three days later deleted a row the very next accept check re-indexed. It is refreshed from the clock before any accept decision in a pass. An invalidated path the index had never held was dropped unread, because it was resolved only through rows the index already had. It now resolves the agent from the same root table discovery reads, and what still cannot be resolved stays queued and counted rather than vanishing. Status told four small lies: a file count that tallied attempts and grew past the total, `current` while work was queued or before any sweep had finished, `current` after close, and a declined read counted as indexed because the parse returned without throwing. It now counts what the store holds, requires all three conditions for `current`, has a `closed` phase, and counts a file only when the index's own cursor moved. The cursor drop ran outside the per-path parse lane, so an overlapping sidebar parse could store its entry in between and turn a forced whole read back into a cache reuse. The decision moves inside the lane as a read requirement, which is the only place that is atomic against it. A sweep retired every indexed file it did not discover. An unmounted volume ENOENTs its whole tree at once, so that deleted a user's searchable history for a detached drive. A root that cannot be read, or that lists nothing where it listed transcripts before, is degraded, and a degraded root's files are never retired however loudly the filesystem says they are gone. The pass loop moves to `session-search-work-loop.ts`: one task at a time, one pending tick, one abort. It is the part with no opinion about transcripts, and the indexer was over the line limit with it inline. * fix(ai-vault-search): count files owed a read once, and pick the ceiling on purpose `filesPending` summed the store's re-read set and the indexer's queue, and a path sits in both the moment a read is declined during a pause and a caller then invalidates the same file. One transcript read as two, with nothing to distinguish that from two transcripts. It is a union by path now. The drop counts stay a sum, because a drop is an event rather than a membership and nothing retains the paths to deduplicate afterwards. The previous commit raised this queue's cap from 2,000 to the store's 20,000 as a side effect of trying to make one number out of two, which the double-count shows it never was. Both queues retain a candidate per entry, so that quietly doubled the worst-case memory of a background feature. Back to 2,000, with the reason written down: the store's set is filled by the reader at machine speed during a pause and needs headroom proportional to the transcripts on the disk, while this one is filled by a cycle's budget rollover, bounded by one recent window at a few hundred, and by `invalidate()`, where 2,000 outstanding requests is already a malfunctioning caller. * fix(ai-vault-search): apply the round-1 guards on the cycle path too Two of the round-1 fixes were written on the sweep and the cycle walked around them twenty seconds later. The degraded-root fence now lives inside the retirement function itself rather than at one call site, so both passes get it from one place and the cycle cannot delete what the sweep just protected. The cycle also reads the sweep's root counts, so it can see a tree that went to zero at all; it does not write them back, because a recent-window discovery is not a census. The N-to-zero alarm was single-shot: the degraded sweep's own zero became the baseline, so the next sweep compared zero with zero and retired the tree it had just spared. A root keeps its last healthy count until one lists it non-empty again. Taking a file no longer means the cursor moved. A forced whole re-read of an unchanged file writes an identical cursor, so an invalidated file that turned out not to have changed was never settled: owed forever, re-read whole every interval, status pinned. It means the index now covers the file at this stat, which is the same question the skip at the top of the pass asks, and it is the same function. An aborted cycle handed the store's re-read set to a queue a tenth its size, which silently discarded the difference. What came from the store goes back to the store, under its own bound and its own retention rule. Whether a sweep finished is now an argument rather than a call site's position, so an aborted one cannot latch `current` by being reported a line too early. * fix(ai-vault-search): fence real directories, and let the alarm release The degraded-root fence did nothing at all for OpenClaw, the one agent whose roots are alternates for a single install. Discovery reports those as one discovery whose rootDir is every path joined by the platform's path delimiter, and that string is not a directory: the probe readdir'd it and got ENOENT, the containment check never matched a file under it, and a scan issue recorded against a real root never compared equal to it. So the agent most likely to live on a mounted volume was the one an unmount deleted, and the degraded root it reported was not a path anyone could act on. Health now runs on the constituent directories, taken from the same source table discovery reads rather than by splitting the joined string back apart, which would be its own bug: a directory may legally contain the delimiter. Files are attributed to the root they actually live under, so one alternate can be unreadable while the other keeps indexing and retiring normally. The N-to-zero alarm also never released. Carrying only counts above zero meant a root the user legitimately emptied stayed degraded for the life of the process, its rows never retired and the phase pinned. The rule is now explicit: a root that cannot be listed keeps its last healthy count and stays degraded indefinitely, while one that lists successfully and empty on two consecutive full sweeps is believed. One sweep is not enough, because that is also what a freshly unmounted volume looks like, and only a sweep counts: a recent-window cycle can see a root at zero but is not a census. The allowance still charges a forced read at the size discovery saw. That is an under-count when a file grows mid-cycle, and it is deliberate: the budget paces a cycle rather than accounting for it, and the error is bounded by what one cycle's writers appended. * fix(ai-vault-search): prove a root once held transcripts from the index, not memory The fence was inert on the first sweep of every process. The evidence that a root had ever held anything lived only in memory, so after a restart it was empty, and a missing directory is what a detached volume and an agent that was never installed both look like. Index a transcript, close, detach the volume, open the same database: every row retired on that one sweep, with no degraded root reported. The evidence now comes from the store, which is the thing that actually outlives the process, through a range scan on the path key. A root that is missing while the index holds files under it is degraded; only one the index holds nothing under is absent. Consecutive also has to mean consecutive. An unreadable sweep left the tally alone rather than breaking it, so empty, unreadable, empty added up to a deletion nobody performed. Anything that is not a successful empty listing now resets the run. Rows under no configured root were immortal: nothing refreshed them, nothing retired them, nothing reported them, and searches still returned them. That happens when a profile moves or a root is reconfigured. One rule, written at the function: such a file is retired exactly like any other if its path answers ENOENT, because that is proof, and otherwise the rows stay and `orphanedFiles` reports them. An index holding content the current configuration cannot reach is a configuration problem to surface, not a licence to delete history. An aborted sweep no longer publishes findings it never gathered. It stops probing on abort, so its empty degraded list would have cleared a live alarm, and its partial view must not count toward emptying a root either. It now skips the health pass entirely and carries the previous state forward. * fix(ai-vault-search): an empty mountpoint is not an emptied root either Round 4 moved the missing-root branch onto the store and left the other one on memory. An unmount on Linux, WSL or sshfs does not remove the mountpoint: it leaves it present and empty, so a detached volume takes the listable-but-empty branch, and that branch armed its two-sweep grace from a count that is zero on the first sweep of every process. Index three transcripts, close, detach: all three retired on that one sweep, with no alarm and a phase of `current`. Both branches now ask the store, which is the only thing that outlives the process. Nothing re-armed a sweep when a root came back. A root absent at start is correctly ignored, but after it returns a cycle only reads the newest N per agent, so one file was indexed and the rest stayed unreachable for the life of the process. A cycle that sees a root listing again where a pass judged it absent or degraded now asks for a sweep. Only after one sweep has completed: before that, a root with no recorded count has simply never been censused, and treating that as a recovery would turn every early cycle into a full sweep. `hasIndexedFilesUnder` was already bounded at a path segment; nothing pinned it, which is why the bare-prefix mutation lived. It has a test now, on both separators, including that a root is not held under itself. * refactor(ai-vault-search): prove a deletion by walking to the root, not by remembering Retirement had grown a root-health state machine: a per-root healthy count, a two-consecutive-empty-sweeps tally, a census flag, a store query for whether the index had ever held files under a root, and a fence every call site had to remember to apply. Four review rounds found the same bug in four shapes, because each shape was a new way for the machine to conclude "empty" from something that was not. The rule is structural now. A row retires only when a directory between the file and its configured root lists successfully and the next component toward the file is absent from that listing; a directory that ENOENTs is walked up, and any other failure is unverifiable at once. The walk stops at the configured root, so everything above it -- a home on an unmounted volume, a detached drive, a dropped SSH mount -- is out of scope by construction rather than by memory, and the rule reads the same on the first pass of a process as on the thousandth. The invariants are written at the top of the module and each is a test. One bit per root survives: a root that held transcripts on the previous pass and holds none on this one gets a pass of grace, so a directory swapped out for a moment cannot retire a tree. What that does not cover is stated in the module and pinned by two tests. degradedRoots becomes a per-pass signal with no memory: roots discovery recorded an issue against, roots the walk could not read through, and roots that yielded nothing and refuse to list at all. * fix(ai-vault-search): close the loop, budget the backfill, and let a pause mean it Five lifecycle defects, all of them cases where a call did more or less than it says. close() disarmed the timer and aborted the task in flight but left the queue running, so a clear() queued a moment earlier would go on to delete the database, open a new one and register a consumer against it, behind an indexer whose caller had finished with it. Closing the work loop makes every queued task a no-op. The backfill was the one pass that read transcript bytes without a budget: a first run over a large disk owned the process until it finished. It now spends an allowance of its own and hands back the rest of its plan, which the passes that follow drain without re-discovering. The allowance is separate from the cycle's and much larger, because a first run has a backlog and steady state does not: 128 MB a pass drains 20 GB in about 53 minutes where the cycle budget would take about 14 hours. A pause now stops purges and compaction too: narrowing the history window while paused records that a purge is owed and runs it on the first pass allowed to write. resume() no longer sweeps on its own, however long the pause was; every read declined while paused is already in the store's re-read set, which the next cycle drains. start() while paused arms on resume instead of queueing a pass that returns immediately and resolves as though one had run. A root that recovers still buys a full sweep, but at most one per recovery: it has to be listed healthy on the pass after the one that re-armed before it can buy another, so a root flapping every interval costs one sweep rather than one a flap. Smaller: clear() resets the swept flag, so an emptied index is not reported as current before the sweep that refills it; a sweep watches only what it could not settle rather than every path it discovered; and the progress pair is measured against one population, so a ratio cannot exceed 100 percent. The round-6 lifecycle matrix lives in the repository now: 11 operations against 3 unreachable-root shapes against both ways discovery reports a root, 66 cells on four invariants. * test(ai-vault-search): drop the tests the old retirement rule owned Two of them asserted the same behaviour as the emptied-root tests that replaced them, and both were named for a rule that no longer exists: retirement waiting for a second sweep to agree, and an unmounted volume being recognised by its root listing empty. Comments that pointed at review rounds rather than at the behaviour go with them, and the merged-root test is named for what it covers now that there is no root-health module for it to be about. * fix(ai-vault-search): stop a sweep from erasing the request that arrived during it Four round-8 findings, all in round-7 code. A full-sweep request raised while a sweep was running was erased by the sweep it arrived during. The flag stayed set across the await and was cleared on the way out, so widening the history window or calling reconcile({ full: true }) part way through a backfill left status reading `current` with the widened-in transcripts never read. The pass takes the flag on entry now; an unfinished sweep is what puts it back. clear() followed by close() left the database on disk. The removal was queued on the work loop, close() makes queued tasks no-ops, and clear()'s promise resolved anyway -- a privacy action that reports success without doing anything. The store, its consumer registration and its file on disk now have one owner and one lifetime, and closing performs a removal that is still owed. The backfill drain bounded bytes but not wall time. The pacer backs off 15 seconds a batch on a loaded host, so a pass could hold the loop for a quarter of an hour without going near its byte budget, and since the drain runs inside the reconcile cycle that is the recent-N-per-interval promise gone. Every read pass now stops at one interval and hands the rest back. A row whose path names an entry inside a container rather than a file of its own was proven only against the container, so an entry deleted inside it could never be retired. Such a row is now proven by the container's own enumeration, under the same bar a directory listing has to meet: exhaustive, successful, and not empty. Nothing in this PR can hold such a row yet -- the index pass refuses a source whose messages the channel cannot reach, which is every OpenCode SQLite session -- so this is the guard for the day that changes, and a test pins the precondition. Also: status() reports `idle` before start() rather than describing work no timer was going to do, and reconcile() before start() is refused rather than writing the index once and leaving it to go stale. * test(ai-vault-search): date the widened-in transcript on the clock retention reads The transcript meant to sit outside a 30-day window was dated against wall time while the window is measured against the test clock, which runs a year behind it, so the file was inside the window and the test proved nothing: it passed with the fix reverted. * fix(ai-vault-search): meet the rewritten store where PR 2 left it The rebase onto the one-transaction-per-file store: re-add the cursor predicate PR 2 dropped, read `sessions`/`messages` now the visibility views are gone, and delete `recoveredRows` -- there is no tombstone table left for a crashed writer to leave rows in, so the counter could only ever read zero. * refactor(ai-vault-search): make the indexer immutable, with one queue and one bound A configuration change is now "close it, construct a new one", so the object has one store, one registration and one lifetime. `pause`, `resume`, `clear`, `setHistoryDays` and `invalidate` are gone, and with them every flag that only existed to keep a second lifetime in step: `paused`, `pausedAt`, `purgeDue`, the deferred-purge path, the resume-sweep rules and the start-while-paused case. Throwing the index away is close, `removeSessionSearchDatabase`, and a new instance; widening retention is a new instance whose opening sweep admits the older files, and narrowing is the purge that opens every full sweep. One queue, not three. The indexer's pending queue and the sweep remainder are deleted; the store's re-read set is the one bounded queue, already the place a declined read lands, and `filesPending` is its size rather than a union across queues that could count one transcript twice. One pacer, not three. The files-and-bytes allowance and the load-average back-off are deleted; a pass reads until its wall-clock deadline and hands the rest back. A pass that runs out of time defers only files it would actually have read, so a truncated pass cannot buy a whole re-read for a file the index already covers. The sweep cadence subsumes root recovery: a full sweep runs on start and every `fullSweepEveryCycles` after it, so a root that comes back is picked up by the next one instead of by a flap-bounded re-arm rule. * test(ai-vault-search): prove close disarms the timer, not only the store Removing `loop.close()` from `close()` failed no test: the unregister already stopped a later scan reaching the index, so the surviving timer and the task queued behind it were invisible. The close test now asserts the timer is gone and that advancing the clock past it reports nothing, which is what a pass running against a shut store would have done. * refactor(ai-vault-search): drop the options and fields nothing reads `retirementChecksPerCycle` had no caller in the stack, so the reconciler keeps its own constant; the error reporter is only needed while the store and the loop are being built, so it stops being a field. * refactor(ai-vault-search): let indexedSources walk the table it is asked for The per-path arm existed for `invalidate()`, which had to resolve a path the index might never have held. Only the sweep reads this now, and it reads all of it. * fix(ai-vault-search): never call a half-written file current PR 2 now reports a file a chunked read left half written with a null cursor under the whole file's mtime and size, so the freshness check has to start at `requiresWholeRead`: comparing only the stat calls a prefix current and leaves it in the index for good. Two consequences, one test each. The skip check no longer skips such a file, and the took-it check no longer reports it indexed, so it stays owed until a read finishes it. The pass reads it whole rather than appending, which repairs it in one pass instead of waiting for the consumer to decline an append it was never going to take. * fix(ai-vault-search): four ways a pass reached the wrong conclusion Round 10, each reproduced on 6a1bcfdf49 first and each repro kept as a test. A sweep watched only what it could not settle, which is nothing on a healthy machine, so the cycle after a sweep had no deletion candidates and the cycle after that no longer remembered the file. A transcript deleted in that interval survived until the next periodic sweep, five minutes later. The sweep now seeds the watch set with its own recency window, taken from its own discoveries through the same class discovery selects with, so there is no second spelling of the rule and no second walk of the trees. A transcript the reader cannot open was recorded stale by the consumer on every attempt and re-read every cycle for ever. Three failures at one unchanged stat now hold a file out until that stat moves, which is the only thing that can mean it changed. `failures`, a tally of attempts that climbed without bound, becomes `unreadableFiles`, a gauge of files being held; a non-zero value is degradation, because waiting will not close that gap. `close()` part way through a pass left the pass reading a shut handle and reported three database errors to the owner who asked for the close, and `status()` afterwards opened it again to answer zero files. The pass stops at the cancellation the close raises, and a closed indexer reports what it last knew. `indexedSources` throws rather than answering with an empty list: its caller is a sweep deciding what nothing rediscovered, and an empty answer is the one conclusion an unreadable handle must not reach. A sweep that threw puts its own flag back, so something is still armed to try again. `droppedPending` was a lifetime tally under a doc that promised it meant the queue was incomplete until the next sweep. A completed sweep now clears it, and `bytesIndexed` resets per pass rather than per sweep, so it stops sawtoothing every fifteen cycles. Two indexers on one database both registered with the reader and wrote every transcript twice. The second construction throws. * test(ai-vault-search): prove the three conclusions a broken pass must not reach Three round-10 mutations survived the first pass of tests, all of them the same shape: a pass that failed still reached a verdict, and nothing checked. The drop reset moves into the sweep itself, where a test holding the store can overflow the queue and watch a completed sweep clear it; from the indexer the call was unreachable without twenty thousand files. A sweep whose held-file read throws now has a test that the failure travels rather than being folded into an empty list, and a sweep that threw part way has one that the flag saying a sweep is owed comes back. * feat(ai-vault-search): put what a file still owes on the file's own row Three additive columns on `files`, and the consumer writes them. `state` is 'current', 'due' or 'failed'; `fail_count` and `failed_mtime_ms` are what stop an unreadable transcript being retried on every pass for ever. Schema version 4, so a stale index rebuilds. Every refusal now leaves its record on the row rather than in a map beside it. A declined append is 'due': the index is behind on a span no append reaches, so the next pass reads the file whole. A read that started and did not commit is 'failed', counted, and stamped with the stat it failed at, because a transcript the reader cannot open fails identically every time and only a change to that stat can mean the file itself changed. A path the file table does not name needs no record at all: the next pass reads it because the index holds nothing for it. Deleted with the in-memory set they served: `markStale`, `takeStale`, `pendingFileCount`, `droppedPendingFileCount`, `forgetDroppedPending`, `setAcceptingWrites`, `acceptsCandidate`, `STALE_PATH_LIMIT`. Added: `files()`, `setFileState()`, `stateCounts()`, `retentionCutoff`. The retention gate moves into `beginWrite`, because the consumer observes every read the session list makes and not only the ones the index asked for. The indexer rewrite that consumes this surface is the commit after; this one is kept to the store, the schema and the consumer so PR 2's own final commits can rebase over it. * fix(ai-vault-search): count a failure for a file the index never held The common unreadable transcript is one no read ever got through: a file behind the wrong mode bits fails on its first attempt, so there is no row to count the failure on and it would be read again on every pass for the life of the process. The failure now inserts its own row, holding a zero cursor and no session, which is what "the index holds nothing for this file" already looked like. * refactor(ai-vault-search): make the store the indexer's only memory Design v3. Every question a pass asks between passes is a row in `files`: what is owed a read, what has failed and how often, what the index holds and therefore what may have been deleted, what to report. Two things outlive a pass and are not rows -- the timer, and one bit per root for the retirement walk's grace -- and both are named in the class doc. One loop, four steps. Discover: the only filesystem walk, every root on a sweep and the newest N per agent on a cycle. Decide: the candidate's stat against its row, as one pure function with its own test. Retire: the rows discovery did not return, inside the scope it covered, through the unchanged walk. Report: a `GROUP BY state` over the same rows. `session-search-backfill.ts` and `session-search-reconciler.ts` become one `session-search-pass.ts`, because the two differed only in discovery scope. Deleted with them: the watch set redefined three times, the hold-out map, the `sweptClean` latch, `session-search-indexing-status.ts` and every counter with a rule about when to reset, `session-search-unreadable-files.ts`, and PR 3's addition to `session-search-file-cursor.ts`, which the decide step replaced. Two things the design did not anticipate, both found by its own tests. A cycle lists the newest N per agent, so a backlog outside that window is invisible to it and a first run would have crawled: a pass that runs out of time now asks for a sweep, which is self-limiting because the first pass that finishes its reads hands the interval back. And a cycle's retirement candidates are the newest rows under the roots it listed, capped, so a deletion inside the recency window is proven on the next cycle whenever it happened rather than waiting for a sweep. * test(ai-vault-search): make the retirement cap test prove its ordering Removing the newest-first sort from a cycle's retirement scope failed nothing: the fixture wrote its transcripts oldest first and the sweep indexed them newest first, so the table's own row order already put the oldest last and an unsorted slice happened to reach the same answer. The oldest file is now indexed on its own first, which makes it the earliest row as well as the oldest file, and the two orders disagree. * fix(ai-vault-search): take schema version 5 for the three files columns PR 2's final pass took version 4 when it dropped conversation_fts, and the rebase merged both bumps into one number. The columns take 5. Two smaller things the same rebase left behind: the stub store in the identity test still declared the two methods the consumer no longer calls, and STALE_PATH_LIMIT outlived the set it bounded. This sits one commit above the columns rather than inside them, because the rebase onto abeccc5905 was the one history rewrite that was authorised. Whoever cherry-picks the store commit needs both. * refactor(ai-vault-search): drop the file count nothing reads, and say what memory is left `store.indexedFileCount` has no caller: the status reports the state counts and PR 4 reads sessions. The class doc claimed two things outlive a pass. Six do, and each is now named with the reason it cannot be a row: the grace bit, the two timer fields, the three the last pass observed for `status()` to answer between passes, and one cached query result read only after a close. A slogan that undercounts is worse than a list, because the next round has to rediscover what it left out. * fix(ai-vault-search): release the database path when the open throws The claim on a database path was staked before the store opened it, and a construction that throws has no close() to release it. One failed open -- a directory where the file should be, a corrupt header, a permission -- left the path owned by an object that does not exist, and every later construction was refused for the life of the process, including the one that would have fixed whatever broke the open. * fix(ai-vault-search): record a source no read can ever index An OpenCode SQLite session decodes where the message channel cannot reach it, so no read of one will ever commit a row, and the consumer declined it without writing anything. That left the file table silent about a source discovery returns on every pass: the decide step saw a path the index held nothing for and asked for a read, and asking for one over a warm cache drops the session list's own resume point. Every OpenCode session was fully decoded on every pass, and the sidebar's cached fold was thrown away with it -- the cache STA-1278 and STA-1417 added. The decline now writes the row the store already has a shape for: a read that went through and decoded no session, cursor at the file's size, no session row. The decide step skips it until its stat moves, and the retirement walk retires it like any other row. Nothing in the decide step knows what OpenCode is. * fix(ai-vault-search): bound the retirement walk by directories, not by rows A directory that cannot be listed answers `unverifiable` for every row under it, on every pass, for as long as the permission stays wrong. Counting rows against the cap let five hundred such rows spend the whole budget on one readdir's worth of verdicts: a row for a file the user really deleted, sorted behind them, was never reached on any pass. Six full sweeps and it was still held; at a hundred rows the same file retires on the first. The cap now counts the directories a walk asks for, which is what actually costs a read, and it is spent only by a row that starts somewhere the walk has not been. Rows sharing a directory are one read and then map lookups, so a block of them can no longer crowd out anything. The comment claiming the leftovers "keep being watched" went with it. There is no watch set: a row the walk did not reach is simply still undiscovered on the next pass, which is what makes finishing over several passes safe. * fix(ai-vault-search): normalize indexer ownership paths --- .../ai-vault-search/session-search-clock.ts | 24 + .../session-search-degraded-roots.ts | 88 ++ .../session-search-deleted-sources.test.ts | 356 ++++++ .../session-search-deleted-sources.ts | 263 ++++ .../session-search-directory-listings.test.ts | 61 + .../session-search-directory-listings.ts | 70 ++ .../session-search-file-write.test.ts | 17 +- .../session-search-index-consumer.test.ts | 85 +- .../session-search-index-consumer.ts | 66 +- .../session-search-index-pass.test.ts | 194 +++ .../session-search-index-pass.ts | 84 ++ .../session-search-index-writer.test.ts | 6 +- .../session-search-indexer-options.ts | 49 + .../session-search-indexer-test-fixture.ts | 168 +++ .../session-search-indexer.test.ts | 1090 +++++++++++++++++ .../ai-vault-search/session-search-indexer.ts | 325 +++++ .../session-search-lifecycle-matrix.test.ts | 378 ++++++ .../session-search-live-transcript.test.ts | 12 +- .../session-search-merged-roots.test.ts | 135 ++ ...ession-search-native-chat-indexing.test.ts | 113 ++ .../session-search-opencode-decline.test.ts | 177 +++ .../ai-vault-search/session-search-pass.ts | 216 ++++ .../session-search-read-decision.test.ts | 117 ++ .../session-search-read-decision.ts | 100 ++ .../session-search-retention-policy.test.ts | 26 + .../session-search-retention-policy.ts | 25 + .../session-search-scan-roots.test.ts | 58 + .../session-search-scan-roots.ts | 135 ++ .../ai-vault-search/session-search-schema.ts | 13 +- .../session-search-store-is-memory.test.ts | 265 ++++ .../ai-vault-search/session-search-store.ts | 199 +-- .../session-search-synthetic-sources.ts | 56 + .../session-search-work-loop.ts | 87 ++ .../ai-vault/session-scanner-parse-cache.ts | 57 +- 34 files changed, 4945 insertions(+), 170 deletions(-) create mode 100644 src/main/ai-vault-search/session-search-clock.ts create mode 100644 src/main/ai-vault-search/session-search-degraded-roots.ts create mode 100644 src/main/ai-vault-search/session-search-deleted-sources.test.ts create mode 100644 src/main/ai-vault-search/session-search-deleted-sources.ts create mode 100644 src/main/ai-vault-search/session-search-directory-listings.test.ts create mode 100644 src/main/ai-vault-search/session-search-directory-listings.ts create mode 100644 src/main/ai-vault-search/session-search-index-pass.test.ts create mode 100644 src/main/ai-vault-search/session-search-index-pass.ts create mode 100644 src/main/ai-vault-search/session-search-indexer-options.ts create mode 100644 src/main/ai-vault-search/session-search-indexer-test-fixture.ts create mode 100644 src/main/ai-vault-search/session-search-indexer.test.ts create mode 100644 src/main/ai-vault-search/session-search-indexer.ts create mode 100644 src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts create mode 100644 src/main/ai-vault-search/session-search-merged-roots.test.ts create mode 100644 src/main/ai-vault-search/session-search-native-chat-indexing.test.ts create mode 100644 src/main/ai-vault-search/session-search-opencode-decline.test.ts create mode 100644 src/main/ai-vault-search/session-search-pass.ts create mode 100644 src/main/ai-vault-search/session-search-read-decision.test.ts create mode 100644 src/main/ai-vault-search/session-search-read-decision.ts create mode 100644 src/main/ai-vault-search/session-search-retention-policy.test.ts create mode 100644 src/main/ai-vault-search/session-search-retention-policy.ts create mode 100644 src/main/ai-vault-search/session-search-scan-roots.test.ts create mode 100644 src/main/ai-vault-search/session-search-scan-roots.ts create mode 100644 src/main/ai-vault-search/session-search-store-is-memory.test.ts create mode 100644 src/main/ai-vault-search/session-search-synthetic-sources.ts create mode 100644 src/main/ai-vault-search/session-search-work-loop.ts diff --git a/src/main/ai-vault-search/session-search-clock.ts b/src/main/ai-vault-search/session-search-clock.ts new file mode 100644 index 00000000000..c9eaf609a34 --- /dev/null +++ b/src/main/ai-vault-search/session-search-clock.ts @@ -0,0 +1,24 @@ +// Why injected rather than the globals: every freshness guarantee this indexer +// makes is "within one reconcile interval", and a guarantee stated in wall time +// is only a claim until a test can advance the clock and watch it hold. + +/** Opaque to the indexer; a fake clock hands back whatever it likes. */ +export type SessionSearchTimerHandle = object | number + +export type SessionSearchClock = { + now(): number + setTimeout(callback: () => void, ms: number): SessionSearchTimerHandle + clearTimeout(handle: SessionSearchTimerHandle): void +} + +export const systemSessionSearchClock: SessionSearchClock = { + now: () => Date.now(), + setTimeout: (callback, ms) => { + const timer = setTimeout(callback, ms) + // Nothing here should hold the process open: the index is a cache, and a + // pending reconcile is never a reason to keep a CLI or a child alive. + timer.unref?.() + return timer + }, + clearTimeout: (handle) => clearTimeout(handle as NodeJS.Timeout) +} diff --git a/src/main/ai-vault-search/session-search-degraded-roots.ts b/src/main/ai-vault-search/session-search-degraded-roots.ts new file mode 100644 index 00000000000..0432cdbc99f --- /dev/null +++ b/src/main/ai-vault-search/session-search-degraded-roots.ts @@ -0,0 +1,88 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import type { SessionSearchDirectoryReader } from './session-search-directory-listings' + +/** A scan root this pass could not read through, and what stopped it. */ +export type SessionSearchDegradedRoot = { root: string; reason: string } + +/** + * Roots a pass could not read, derived from that pass alone. + * + * There is no root-health state machine any more and nothing is carried between + * passes: "degraded" now means one of two things this pass observed, both of + * which are readdir results. + * + * 1. Discovery recorded a scan issue against the root itself — a stalled WSL + * distro, a gate refusal, an unreadable tree. + * 2. The retirement walk could not prove a file the index holds under that root + * either present or gone, because a directory between the file and the root + * refused to list, or because the root itself is not there. + * 3. A root that yielded no transcripts refuses to list at all. The file walker + * swallows a readdir failure and returns, so without this an EACCES root and + * an agent that was never installed both arrive as "no files" — reporting + * the first as an empty index is the loss-of-contact-as-absence mistake + * docs/reference/ssh-execution-boundary.md forbids. + * + * The second is what reports a detached volume, and it needs no memory of + * previous passes: the evidence is the index's own rows plus this pass's + * readdir errors. A root the index holds nothing under and cannot list is + * reported by the third; a root that is simply missing is not reported at all, + * because that is what an agent nobody installed looks like. + */ +export function scanIssueDegradedRoots( + roots: readonly string[], + issues: readonly AiVaultScanIssue[] +): SessionSearchDegradedRoot[] { + const degraded = new Map() + for (const issue of issues) { + // 'notice' rows are scanner commentary; a per-file failure is not a root's. + if (issue.kind !== 'notice' && roots.includes(issue.path)) { + degraded.set(issue.path, issue.message) + } + } + return [...degraded].map(([root, reason]) => ({ root, reason })) +} + +/** One entry per root, first reason kept, so a pass reports each root once. */ +export function mergeDegradedRoots( + ...groups: readonly (readonly SessionSearchDegradedRoot[])[] +): SessionSearchDegradedRoot[] { + const merged = new Map() + for (const group of groups) { + for (const degraded of group) { + if (!merged.has(degraded.root)) { + merged.set(degraded.root, degraded.reason) + } + } + } + return [...merged].map(([root, reason]) => ({ root, reason })) +} + +// A missing root is not a broken one: an uninstalled agent's root answers +// exactly this, and the index holding rows under it is what the retirement +// walk reports instead. +const MISSING_ROOT = new Set(['ENOENT', 'ENOTDIR']) + +/** + * Roots that yielded no transcripts and cannot be listed either. + * + * Only roots a pass found empty are read: one that returned files is readable + * by construction. The read shares the pass's listing cache, so a root the + * retirement walk also has to ask about costs one readdir between them. + */ +export async function unreadableRoots( + roots: readonly string[], + listings: SessionSearchDirectoryReader, + signal?: AbortSignal +): Promise { + const degraded: SessionSearchDegradedRoot[] = [] + for (const root of roots) { + if (signal?.aborted) { + break + } + const listing = await listings.namesIn(root, signal) + if (!listing.listed && !(listing.code !== null && MISSING_ROOT.has(listing.code))) { + degraded.push({ root, reason: listing.message }) + } + } + return degraded +} diff --git a/src/main/ai-vault-search/session-search-deleted-sources.test.ts b/src/main/ai-vault-search/session-search-deleted-sources.test.ts new file mode 100644 index 00000000000..d1c367d632f --- /dev/null +++ b/src/main/ai-vault-search/session-search-deleted-sources.test.ts @@ -0,0 +1,356 @@ +import { chmod, mkdir, rm, writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { parserPublishesMessages } from '../ai-vault/session-scanner-agent-parser' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { retireDeletedSessionSearchSources } from './session-search-deleted-sources' +import { + SessionSearchDirectoryListings, + type SessionSearchDirectoryListing, + type SessionSearchDirectoryReader +} from './session-search-directory-listings' +import { + openSessionSearchIndexerHarness, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' +import { SessionSearchStore } from './session-search-store' + +// The invariants this file exists to pin are written at the top of +// session-search-deleted-sources.ts. Each one is named in the tests below. + +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 + +let harness: SessionSearchIndexerHarness +let store: SessionSearchStore +let removed: string[] + +beforeEach(async () => { + resetTranscriptConsumersForTests() + harness = await openSessionSearchIndexerHarness('ss-deleted-sources') + removed = [] + store = new SessionSearchStore(harness.databasePath) + // Only the removal matters here; the store's own removal path has its own tests. + store.removeFile = (path: string) => removed.push(path) +}) + +afterEach(async () => { + store.close() + await harness.cleanup() +}) + +/** A reader that answers with whatever a stalled mount would, per directory. */ +function readerAnswering( + answers: Record +): SessionSearchDirectoryReader { + return { + namesIn: (directory) => + Promise.resolve( + answers[directory] ?? { listed: false, code: 'ENOENT', message: 'no such directory' } + ) + } +} + +function retire( + paths: readonly string[], + options: { + roots?: readonly string[] + emptiedRoots?: ReadonlySet + enumeratedContainers?: ReadonlyMap> + listings?: SessionSearchDirectoryReader + directoryLimit?: number + } = {} +) { + return retireDeletedSessionSearchSources({ + store, + paths, + roots: options.roots ?? [harness.roots.claudeProjectsDir ?? ''], + emptiedRoots: options.emptiedRoots, + enumeratedContainers: options.enumeratedContainers, + listings: options.listings ?? new SessionSearchDirectoryListings(), + directoryLimit: options.directoryLimit + }) +} + +// I4: a file the user deleted retires on the first pass that proves it, with no +// waiting period, because its directory listed and it was not in the listing. +it('retires a deleted file the moment its own directory lists without it', async () => { + const kept = join(harness.claudeProjectDir, 'kept.jsonl') + await mkdir(harness.claudeProjectDir, { recursive: true }) + await writeFile(kept, '{}') + const deleted = join(harness.claudeProjectDir, 'deleted.jsonl') + + const result = await retire([kept, deleted]) + expect(result.retired).toEqual([deleted]) + expect(removed).toEqual([deleted]) + // A file that is still there is settled, not watched: it is neither retired + // nor carried into the next pass as unfinished business. + expect(result.unverifiable).toEqual([]) + expect(result.degradedRoots).toEqual([]) +}) + +// I4, the other shape: the directory itself is gone, so the question moves up +// one level and the root answers it. +it('retires a whole project directory the user deleted', async () => { + const sibling = join(harness.roots.claudeProjectsDir ?? '', 'other', 'kept.jsonl') + await mkdir(join(harness.roots.claudeProjectsDir ?? '', 'other'), { recursive: true }) + await writeFile(sibling, '{}') + const gone = join(harness.claudeProjectDir, 'inside-a-deleted-project.jsonl') + + const result = await retire([gone]) + expect(result.retired).toEqual([gone]) + expect(result.unverifiable).toEqual([]) +}) + +// I1 and I2: a root that is not there proves nothing. The walk stops at the +// configured root and never asks what is above it, so a home directory on an +// unmounted volume — the shape a detached drive or a dropped SSH mount takes — +// leaves every row exactly where it was. +it('keeps every row under a root that is not there', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + const held = [join(harness.claudeProjectDir, 'one.jsonl'), join(root, 'flat.jsonl')] + + const result = await retire(held) + expect(result.retired).toEqual([]) + expect(result.unverifiable).toEqual(held) + // The root is named, once, so a caller can say which tree is unreachable. + expect(result.degradedRoots).toEqual([{ root, reason: `${root} could not be listed.` }]) +}) + +// I3: the same answer with no memory at all. Nothing here is carried from a +// previous pass, which is what makes the first sweep after a restart — when a +// volume is most likely to be missing — behave like every other pass. +it('keeps a missing root on a pass that has seen nothing before it', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + const held = join(harness.claudeProjectDir, 'one.jsonl') + const first = await retire([held], { emptiedRoots: new Set() }) + const second = await retire([held], { emptiedRoots: new Set() }) + expect([first.retired, second.retired]).toEqual([[], []]) + expect(second.degradedRoots.map((one) => one.root)).toEqual([root]) +}) + +// I2: an unreadable directory is not an empty one. EACCES stops the walk where +// it is rather than being walked up like a missing component. +it.skipIf(!CAN_DENY_READ)('keeps rows under a directory that refuses to list', async () => { + const blocked = join(harness.roots.claudeProjectsDir ?? '', 'blocked') + await mkdir(blocked, { recursive: true }) + const hidden = join(blocked, 'hidden.jsonl') + await writeFile(hidden, '{}') + await chmod(blocked, 0o000) + try { + const result = await retire([hidden]) + expect(result.retired).toEqual([]) + expect(result.unverifiable).toEqual([hidden]) + expect(result.degradedRoots.map((one) => one.root)).toEqual([harness.roots.claudeProjectsDir]) + } finally { + await chmod(blocked, 0o755) + } +}) + +// I2, without needing a filesystem that can produce it: a stalled network mount +// answers EIO or a WSL gate refusal, and neither is ENOENT. This is the SSH and +// WSL case — loss of contact is never evidence of absence. +it('keeps rows when a directory answers with a transport failure', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + const held = join(harness.claudeProjectDir, 'one.jsonl') + for (const listing of [ + { listed: false as const, code: 'EIO', message: 'input/output error' }, + { listed: false as const, code: 'ETIMEDOUT', message: 'the mount stopped answering' }, + { listed: false as const, code: null, message: 'The distro stopped responding.' } + ]) { + const result = await retire([held], { + listings: readerAnswering({ [harness.claudeProjectDir]: listing }) + }) + expect(result.retired).toEqual([]) + expect(result.degradedRoots).toEqual([{ root, reason: listing.message }]) + } +}) + +// The one bit of memory, and the only thing it buys: a root that held +// transcripts on the previous pass and lists empty on this one gets one pass of +// grace, so a directory swapped out for a moment cannot retire a tree. +it('holds a root that went from holding transcripts to empty in one pass', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + await mkdir(root, { recursive: true }) + const held = join(harness.claudeProjectDir, 'one.jsonl') + + const grace = await retire([held], { emptiedRoots: new Set([root]) }) + expect(grace.retired).toEqual([]) + expect(grace.unverifiable).toEqual([held]) + + // The next pass has no transition to point at, so the empty listing is what + // it says it is: the user emptied the root. + const after = await retire([held], { emptiedRoots: new Set() }) + expect(after.retired).toEqual([held]) +}) + +// A flat-layout agent, where the mountpoint IS the session directory, is the +// one shape the grace exists for: there is no intermediate directory whose +// absence could stop the walk. +it('holds a flat root that emptied in one pass, and retires it on the next', async () => { + const root = harness.roots.copilotSessionsDir ?? '' + await mkdir(root, { recursive: true }) + const held = join(root, 'session.jsonl') + + expect((await retire([held], { roots: [root], emptiedRoots: new Set([root]) })).retired).toEqual( + [] + ) + expect((await retire([held], { roots: [root] })).retired).toEqual([held]) +}) + +// OpenClaw's discovery merges two directories into one delimiter-joined label. +// Roots reach this function as the real directories behind that label, so one +// of them being unreachable never touches the other's rows. +it('judges each merged-root directory on its own', async () => { + const current = join(harness.roots.openclawStateDir ?? '', 'agents') + const legacy = join(harness.roots.openclawLegacyStateDir ?? '', 'agents') + const onMissing = join(current, 'main', 'sessions', 'mounted.jsonl') + const deleted = join(legacy, 'main', 'sessions', 'deleted.jsonl') + await mkdir(join(legacy, 'main', 'sessions'), { recursive: true }) + + const result = await retire([onMissing, deleted], { roots: [current, legacy] }) + expect(result.retired).toEqual([deleted]) + expect(result.unverifiable).toEqual([onMissing]) + expect(result.degradedRoots.map((one) => one.root)).toEqual([current]) +}) + +// A row under no configured root is judged by its own directory and nothing +// above it, so a moved profile is never retired on the strength of a root that +// no longer covers it. +it('judges a row under no configured root by its own directory', async () => { + const orphanDir = join(harness.root, 'moved-profile') + await mkdir(orphanDir, { recursive: true }) + const gone = join(orphanDir, 'gone.jsonl') + const present = join(orphanDir, 'present.jsonl') + await writeFile(present, '{}') + + const result = await retire([gone, present], { roots: [] }) + expect(result.retired).toEqual([gone]) + // No configured root owns it, so nothing is reported as degraded for it. + expect(result.degradedRoots).toEqual([]) +}) + +// I8. A synthetic row names a container and an entry inside it. Walking the +// row's own path would report every one of them gone, and walking only the +// container proves nothing about the entry: a session deleted inside a database +// that is still there would never be retired at all. +it('proves a synthetic row against its container, not against its own path', async () => { + const db = join(harness.root, 'opencode.db') + await writeFile(db, '') + const kept = `${db}#session-1` + const deleted = `${db}#session-2` + const enumeratedContainers = new Map([[db, new Set(['session-1'])]]) + + const result = await retire([kept, deleted], { roots: [], enumeratedContainers }) + expect(result.retired).toEqual([deleted]) + expect(result.unverifiable).toEqual([]) +}) + +it('keeps a synthetic row when this pass did not enumerate its container', async () => { + const db = join(harness.root, 'opencode.db') + await writeFile(db, '') + const row = `${db}#session-1` + + // A cycle asks for the newest N per agent, so a row it did not return may be + // the one after them. It enumerates nothing and therefore proves nothing. + await expect(retire([row], { roots: [] })).resolves.toMatchObject({ + retired: [], + unverifiable: [row] + }) + + // An enumeration that returned nothing at all is not evidence either: a + // database whose schema this scanner no longer recognises reads as empty + // with no error, and believing it would retire every session in one pass. + await expect( + retire([row], { roots: [], enumeratedContainers: new Map([[db, new Set()]]) }) + ).resolves.toMatchObject({ retired: [], unverifiable: [row] }) +}) + +it('retires a synthetic row when the container it came from is gone', async () => { + const db = join(harness.root, 'opencode.db') + await writeFile(db, '') + const row = `${db}#session-1` + const enumeratedContainers = new Map([[db, new Set(['session-1'])]]) + await expect(retire([row], { roots: [], enumeratedContainers })).resolves.toMatchObject({ + retired: [] + }) + + await rm(db) + await expect(retire([row], { roots: [], enumeratedContainers })).resolves.toMatchObject({ + retired: [row] + }) +}) + +// Nothing in this PR can hold a synthetic row: the index pass refuses a source +// whose parser decodes its messages where the message channel cannot reach +// them, and OpenCode's SQLite sessions are read on a worker thread. The rule +// above is the guard for the day that changes -- without it the walk would read +// `#` as a filename and retire every such row the moment it appeared. +it('does not index a source whose messages the channel cannot reach', () => { + const db = join(harness.root, 'opencode.db') + expect( + parserPublishesMessages({ + agent: 'opencode', + codexHome: null, + file: { path: `${db}#session-1`, mtimeMs: 1, modifiedAt: '', sizeBytes: 0 } + }) + ).toBe(false) +}) + +// Round 12, F1. The cap counts directories because that is what costs: rows +// sharing one are a single read and then map lookups. +it('caps the directories one pass reads, not the rows it answers', async () => { + const roots = [harness.claudeProjectDir] + const inside = (folder: string, name: string): string => + join(harness.claudeProjectDir, folder, name) + for (const folder of ['one', 'two', 'three']) { + await mkdir(join(harness.claudeProjectDir, folder), { recursive: true }) + } + // Four rows in each of three directories: three reads, twelve answers. + const paths = ['one', 'two', 'three'].flatMap((folder) => + ['a', 'b', 'c', 'd'].map((name) => inside(folder, name)) + ) + + const result = await retire(paths, { roots, directoryLimit: 2 }) + + // Two directories' worth answered, all eight of their rows, and the third + // directory's four left for the pass after this one. + expect(result.retired).toEqual(paths.slice(0, 8)) + expect(result.unchecked).toEqual(paths.slice(8)) +}) + +// The starvation this replaced: an unreadable directory answers `unverifiable` +// for every row under it and never becomes readable, so a cap on rows let one +// such directory hold the walk for as long as the permission stayed wrong. +it.skipIf(!CAN_DENY_READ)( + 'is not starved by many rows under one unreadable directory', + async () => { + const locked = join(harness.claudeProjectDir, 'locked') + await mkdir(locked, { recursive: true }) + const blocked = Array.from({ length: 520 }, (_unused, index) => + join(locked, `locked-${index}.jsonl`) + ) + const deleted = join(harness.claudeProjectDir, 'deleted.jsonl') + await chmod(locked, 0o000) + try { + const result = await retire([...blocked, deleted], { directoryLimit: 512 }) + + expect(result.retired).toEqual([deleted]) + expect(result.unverifiable).toHaveLength(blocked.length) + expect(result.unchecked).toEqual([]) + } finally { + await chmod(locked, 0o700) + } + } +) + +it('reads each directory once however many files it is asked about', async () => { + await mkdir(harness.claudeProjectDir, { recursive: true }) + const listings = new SessionSearchDirectoryListings() + await retire( + Array.from({ length: 50 }, (_unused, index) => + join(harness.claudeProjectDir, `gone-${index}.jsonl`) + ), + { listings } + ) + expect(listings.size).toBe(1) +}) diff --git a/src/main/ai-vault-search/session-search-deleted-sources.ts b/src/main/ai-vault-search/session-search-deleted-sources.ts new file mode 100644 index 00000000000..e600cd3c4b1 --- /dev/null +++ b/src/main/ai-vault-search/session-search-deleted-sources.ts @@ -0,0 +1,263 @@ +import { basename, dirname } from 'node:path' +import type { SessionSearchDegradedRoot } from './session-search-degraded-roots' +import type { SessionSearchDirectoryReader } from './session-search-directory-listings' +import { isUnderScanRoot } from './session-search-scan-roots' +import { splitSyntheticSessionSource } from './session-search-synthetic-sources' +import type { SessionSearchStore } from './session-search-store' + +/* + * Retirement invariants. Every one of these is a test; changing this file means + * changing the list, not working around it. + * + * I1. A row is retired only when its file is PROVEN gone: some directory + * between the file and its configured root lists successfully, and the next + * path component toward the file is absent from that listing. + * I2. If no directory from the file's parent up to the configured root can be + * listed, nothing is proven and no row is dropped. ENOENT/ENOTDIR is walked + * up (the directory itself is a missing component of some ancestor); + * EACCES, EIO, a WSL gate refusal, anything else, is unverifiable at once. + * I3. The rule is the same on the first pass after a process start and on every + * later pass. It needs no memory of what previous passes saw, because the + * walk is bounded at the configured root and never reasons about what is + * above it. + * I4. A file, or a project directory, the user really deleted retires on the + * first pass that proves it. There is no waiting period and no census. + * I8. A row whose path names an entry inside a container rather than a file of + * its own is proven the same way, one level up: the container must be + * present, and the pass must have enumerated it in full and successfully. + * A listing is a listing whether it comes from readdir or from a database. + * + * What I3 costs, stated rather than hidden: a volume mounted at exactly a + * configured root, unmounted so that the mountpoint stays present and lists + * empty, is indistinguishable from a root the user emptied. It retires. The + * realistic unmount shapes do not: a mount above the root leaves the root + * itself missing (the walk stops at the root boundary), and an unreadable root + * is an error, not a listing. One bit per root buys the remaining grace: a root + * that held transcripts on the previous pass and holds none on this one is + * unverifiable for that pass, so a single flap cannot retire a tree. + */ + +// Walked up rather than believed: a directory that ENOENTs is itself the +// missing component its parent has to be asked about. +const MISSING_DIRECTORY = new Set(['ENOENT', 'ENOTDIR']) + +export type SessionSearchRetirement = { + /** Paths proven gone and dropped from the index. */ + retired: string[] + /** Rows kept: this pass could prove the file neither present nor gone. */ + unverifiable: string[] + /** Paths the per-pass cap left for next time. */ + unchecked: string[] + /** Roots owning at least one unverifiable verdict, with the reason. */ + degradedRoots: SessionSearchDegradedRoot[] +} + +export type SessionSearchRetirementArgs = { + store: SessionSearchStore + /** Held paths this pass did not discover; everything else is still there. */ + paths: readonly string[] + /** The real directories this pass walked; the longest one containing a path bounds its walk. */ + roots: readonly string[] + /** Roots that listed transcripts on the previous pass and none on this one. */ + emptiedRoots?: ReadonlySet + /** + * Containers this pass enumerated in full, with the ids each holds. Only a + * census builds it; see session-search-synthetic-sources.ts for the bar a + * container has to meet before it appears here. + */ + enumeratedContainers?: ReadonlyMap> + /** One readdir per directory per pass, shared with the rest of the pass. */ + listings: SessionSearchDirectoryReader + /** + * Directories this walk may read before the pass moves on. + * + * Directories, not rows. A row whose walk finds its directory already read is + * answered from the pass's cache and costs nothing, so counting rows made an + * unreadable directory able to starve the whole walk: five hundred rows under + * one EACCES directory are one readdir and five hundred identical + * unverifiable verdicts, and a row for a file the user really deleted, sorted + * behind them, was never reached on any pass. + */ + directoryLimit?: number + signal?: AbortSignal +} + +type SessionSearchSourceVerdict = + | { verdict: 'gone' } + | { verdict: 'present' } + | { verdict: 'unverifiable'; reason: string } + +/** + * Retires index rows for sources that are provably gone. + * + * One function, called by both the sweep and the cycle, because either one + * alone deleting a user's history the first time a mount is missing is the bug + * this feature kept shipping. There is no separate root fence: the walk cannot + * reach a verdict of `gone` without a successful listing, so an unreadable or + * missing root produces `unverifiable` structurally rather than by a guard + * somebody has to remember to call (docs/reference/ssh-execution-boundary.md: + * loss of contact is never evidence of absence). + */ +export async function retireDeletedSessionSearchSources( + args: SessionSearchRetirementArgs +): Promise { + const { store, paths, signal } = args + const emptiedRoots = args.emptiedRoots ?? new Set() + const directoryLimit = args.directoryLimit ?? Number.POSITIVE_INFINITY + // Every directory this walk asked for, whether the pass had already read it + // or not. What it bounds is real work: a repeat of one already in here is a + // map lookup, and only a name that is new to it can cost a readdir. + const asked = new Set() + const listings: SessionSearchDirectoryReader = { + namesIn: (directory, signal) => { + asked.add(directory) + return args.listings.namesIn(directory, signal) + } + } + const retirement: SessionSearchRetirement = { + retired: [], + unverifiable: [], + unchecked: [], + degradedRoots: [] + } + const degraded = new Map() + for (const [index, path] of paths.entries()) { + // A synthetic row names a container and an entry inside it, never a file of + // its own; walking the row's own path would report every one of them gone. + const synthetic = splitSyntheticSessionSource(path) + const filePath = synthetic?.container ?? path + // Why capped at all: the sweep hands over every path it holds and did not + // discover, and under an unmount that is the whole index. What is left is + // simply still undiscovered next pass, so the walk finishes over the ones + // that follow rather than holding this one. + // + // Spent past the bound only by a row that starts somewhere new. One this + // walk has already read is answered from the map, so refusing it would buy + // nothing and would leave the budget hostage to whichever directory the + // rows happened to be sorted by. + if (signal?.aborted || (asked.size >= directoryLimit && !asked.has(dirname(filePath)))) { + retirement.unchecked.push(...paths.slice(index)) + break + } + const root = configuredRootFor(filePath, args.roots) + const containerProof = await proveSource(filePath, root ?? dirname(filePath), { + listings, + emptiedRoots, + signal + }) + const proof = synthetic + ? proveSyntheticSource(synthetic, containerProof, args.enumeratedContainers) + : containerProof + if (proof.verdict === 'gone') { + store.removeFile(path) + retirement.retired.push(path) + continue + } + if (proof.verdict === 'present') { + continue + } + retirement.unverifiable.push(path) + // Only a configured root is an alarm worth raising: a row under no root + // this scan walks is already reported on its own, as an orphan. + if (root !== null && !degraded.has(root)) { + degraded.set(root, proof.reason) + } + } + retirement.degradedRoots = [...degraded].map(([root, reason]) => ({ root, reason })) + return retirement +} + +/** + * Walks from the file toward its configured root, asking each directory whether + * the next component toward the file is there. The first directory that answers + * decides; a directory that is itself missing moves the question up one level. + * + * The loop cannot pass the configured root, which is what makes the whole thing + * memoryless: everything above the root — a home directory on an unmounted + * volume, a detached drive, an SSH mount that is not there — is out of scope by + * construction rather than by a state machine that has to remember it. + */ +async function proveSource( + path: string, + root: string, + context: { + listings: SessionSearchDirectoryReader + emptiedRoots: ReadonlySet + signal?: AbortSignal + } +): Promise { + let directory = dirname(path) + let child = basename(path) + while (directory === root || isUnderScanRoot(directory, root)) { + const listing = await context.listings.namesIn(directory, context.signal) + if (!listing.listed) { + if (listing.code !== null && MISSING_DIRECTORY.has(listing.code)) { + const parent = dirname(directory) + if (parent === directory) { + break + } + child = basename(directory) + directory = parent + continue + } + return { verdict: 'unverifiable', reason: listing.message } + } + if (listing.names.has(child)) { + return { verdict: 'present' } + } + if (directory === root && context.emptiedRoots.has(root)) { + // One pass of grace, so a root that blinks empty for a moment — a sync + // client mid-swap, a mount that has not settled — cannot retire a tree. + return { + verdict: 'unverifiable', + reason: 'Listed no transcripts where it listed some on the previous pass.' + } + } + return { verdict: 'gone' } + } + return { verdict: 'unverifiable', reason: `${root} could not be listed.` } +} + +/** + * A synthetic row is proven by its container's own enumeration, one level above + * where the filesystem walk stops. + * + * The container has to be present first: a database on a volume that is not + * there proves nothing about the sessions inside it, and a database that is + * gone takes its sessions with it. Only then does the enumeration decide, and + * only when this pass made one that was exhaustive and successful -- a cycle + * asks for the newest N per agent, so an id it did not return may just be the + * one after them. + */ +function proveSyntheticSource( + synthetic: { container: string; id: string }, + containerProof: SessionSearchSourceVerdict, + enumerated?: ReadonlyMap> +): SessionSearchSourceVerdict { + if (containerProof.verdict !== 'present') { + return containerProof + } + const ids = enumerated?.get(synthetic.container) + // An enumeration that returned nothing at all is not evidence that the + // container holds nothing: a source whose schema this scanner no longer + // recognises reads as empty with no error to see, and believing it would + // retire every entry in one pass. + if (!ids || ids.size === 0) { + return { + verdict: 'unverifiable', + reason: `${synthetic.container} was not enumerated in full this pass.` + } + } + return ids.has(synthetic.id) ? { verdict: 'present' } : { verdict: 'gone' } +} + +/** Longest configured root containing the path, or null for a row under none. */ +function configuredRootFor(path: string, roots: readonly string[]): string | null { + let owner: string | null = null + for (const root of roots) { + if (isUnderScanRoot(path, root) && (owner === null || root.length > owner.length)) { + owner = root + } + } + return owner +} diff --git a/src/main/ai-vault-search/session-search-directory-listings.test.ts b/src/main/ai-vault-search/session-search-directory-listings.test.ts new file mode 100644 index 00000000000..e0143cd7a27 --- /dev/null +++ b/src/main/ai-vault-search/session-search-directory-listings.test.ts @@ -0,0 +1,61 @@ +import { mkdir, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { beforeEach, expect, it, vi } from 'vitest' +import { SessionSearchDirectoryListings } from './session-search-directory-listings' + +const { readdir } = vi.hoisted(() => ({ readdir: vi.fn() })) + +vi.mock('../native-chat/wsl-transcript-fs-access', () => ({ + wslGatedReaddir: readdir +})) + +beforeEach(() => { + readdir.mockReset() +}) + +// A WSL root is a UNC path into the distro, and reading it with raw `fs` is +// what makes a stalled distro look like an empty directory. The gated primitive +// is the same one discovery walks with, so a refusal arrives as an error the +// walk treats as unverifiable rather than as "nothing here". +it('reads through the gated primitive, on the scan lane', async () => { + const unc = '\\\\wsl$\\Ubuntu\\home\\me\\.claude\\projects' + readdir.mockResolvedValueOnce([{ name: 'one.jsonl' }]) + const listings = new SessionSearchDirectoryListings() + + const listing = await listings.namesIn(unc) + + expect(readdir).toHaveBeenCalledWith(unc, 'scan', undefined) + expect(listing).toEqual({ listed: true, names: new Set(['one.jsonl']) }) +}) + +it('reports the code a failed read carried, so ENOENT and EACCES stay apart', async () => { + readdir.mockRejectedValueOnce(Object.assign(new Error('permission denied'), { code: 'EACCES' })) + const listings = new SessionSearchDirectoryListings() + expect(await listings.namesIn('/blocked')).toEqual({ + listed: false, + code: 'EACCES', + message: 'permission denied' + }) +}) + +it('reads a directory once per pass, error or not', async () => { + readdir.mockRejectedValue(Object.assign(new Error('gone'), { code: 'ENOENT' })) + const listings = new SessionSearchDirectoryListings() + await listings.namesIn('/gone') + await listings.namesIn('/gone') + expect(readdir).toHaveBeenCalledTimes(1) + expect(listings.size).toBe(1) +}) + +it('is a real directory read when nothing is mocked out from under it', async () => { + readdir.mockImplementation(async (path: string) => { + const { readdir: real } = await import('node:fs/promises') + return (await real(path, { withFileTypes: true })) as unknown + }) + const root = join(tmpdir(), `ss-listings-${process.pid}`) + await mkdir(root, { recursive: true }) + await writeFile(join(root, 'present.jsonl'), '{}') + const listing = await new SessionSearchDirectoryListings().namesIn(root) + expect(listing.listed && listing.names.has('present.jsonl')).toBe(true) +}) diff --git a/src/main/ai-vault-search/session-search-directory-listings.ts b/src/main/ai-vault-search/session-search-directory-listings.ts new file mode 100644 index 00000000000..f2cb13f9168 --- /dev/null +++ b/src/main/ai-vault-search/session-search-directory-listings.ts @@ -0,0 +1,70 @@ +import { wslGatedReaddir } from '../native-chat/wsl-transcript-fs-access' + +/** One directory read: the names it holds, or what stopped the read. */ +export type SessionSearchDirectoryListing = + | { listed: true; names: ReadonlySet } + | { listed: false; code: string | null; message: string } + +/** + * What the retirement walk needs of a directory: its names, or why not. + * + * An interface rather than the class, so a test can hand the walk an EIO or a + * gate refusal — the shapes a stalled network mount answers with, which no + * temporary directory can be made to produce. + */ +export type SessionSearchDirectoryReader = { + namesIn(directory: string, signal?: AbortSignal): Promise +} + +/** + * Every directory one pass had to read, read once. + * + * The retirement walk asks the same directories about many files — a project + * directory holds hundreds of transcripts — and under an unmount every path + * under a root walks up through the same ancestors. One readdir per directory + * per pass keeps that bounded, and it also makes the pass self-consistent: two + * files in one directory cannot get contradictory verdicts because the + * directory changed between them. + * + * Reads go through the same gated primitive discovery uses, so a WSL UNC path + * is routed to the distro's helper process rather than read with raw fs, and a + * gate refusal arrives as an error rather than as an empty directory. + */ +export class SessionSearchDirectoryListings implements SessionSearchDirectoryReader { + private readonly listings = new Map() + + async namesIn(directory: string, signal?: AbortSignal): Promise { + const cached = this.listings.get(directory) + if (cached) { + return cached + } + const listing = await readDirectory(directory, signal) + this.listings.set(directory, listing) + return listing + } + + /** Directories read this pass; only tests and cost accounting need it. */ + get size(): number { + return this.listings.size + } +} + +async function readDirectory( + directory: string, + signal?: AbortSignal +): Promise { + try { + const entries = await wslGatedReaddir(directory, 'scan', signal) + return { listed: true, names: new Set(entries.map((entry) => entry.name)) } + } catch (error) { + const code = + error && typeof error === 'object' && 'code' in error && typeof error.code === 'string' + ? error.code + : null + return { + listed: false, + code, + message: error instanceof Error ? error.message : String(error) + } + } +} diff --git a/src/main/ai-vault-search/session-search-file-write.test.ts b/src/main/ai-vault-search/session-search-file-write.test.ts index 942565e18d6..2c0be88b026 100644 --- a/src/main/ai-vault-search/session-search-file-write.test.ts +++ b/src/main/ai-vault-search/session-search-file-write.test.ts @@ -137,8 +137,11 @@ it('rolls a whole file back when a write throws part way through its transaction expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(3) expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(40) expect(errors).toHaveLength(1) - // The file is owed a re-read, which is the only reason anything was lost. - expect(store.takeStale().map((candidate) => candidate.file.path)).toEqual([SYNTHETIC_TRANSCRIPT]) + // The row itself says the read failed, which is the only reason anything was + // lost and the only record that outlives this read. + expect( + index.db.prepare('SELECT state, fail_count FROM files WHERE path = ?').get(SYNTHETIC_TRANSCRIPT) + ).toMatchObject({ state: 'failed', fail_count: 1 }) // And the connection is usable again: a transaction left open by the failure // would take down every write after it, not just the one that threw. @@ -590,10 +593,16 @@ it('writes nothing for an incomplete read and owes the file a whole re-read', () expect(counts(index.db)).toMatchObject({ sessions: 0, messages: 0, - files: 0, full: 0 }) - expect(store.pendingFileCount).toBe(1) + // One row, holding nothing but the failure: an incomplete read indexes no + // content, and the count of how often it has happened at this stat is the + // only thing that stops the file being read again on every pass. + expect(index.db.prepare('SELECT byte_offset, state, fail_count FROM files').get()).toMatchObject({ + byte_offset: 0, + state: 'failed', + fail_count: 1 + }) expect(errors).toEqual([]) }) diff --git a/src/main/ai-vault-search/session-search-index-consumer.test.ts b/src/main/ai-vault-search/session-search-index-consumer.test.ts index ca02333cf0b..ee855409fb7 100644 --- a/src/main/ai-vault-search/session-search-index-consumer.test.ts +++ b/src/main/ai-vault-search/session-search-index-consumer.test.ts @@ -10,7 +10,7 @@ import { userMessages, type SessionSearchIndexFile } from './session-search-index-test-fixture' -import { SessionSearchStore, STALE_PATH_LIMIT } from './session-search-store' +import { SessionSearchStore } from './session-search-store' let index: SessionSearchIndexFile let store: SessionSearchStore @@ -41,6 +41,13 @@ function cursor(): number | null | undefined { return store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset } +/** What the row itself says it still owes, which is the only record there is. */ +function owed(): { state: string; fail_count: number } | undefined { + return index.db + .prepare('SELECT state, fail_count FROM files WHERE path = ?') + .get(SYNTHETIC_TRANSCRIPT) as { state: string; fail_count: number } | undefined +} + it('appends onto its own cursor and carries the content hash forward', async () => { replayTranscriptRead({ messages: userMessages('first half', 3), @@ -64,7 +71,7 @@ it('appends onto its own cursor and carries the content hash forward', async () .get() as { hash: string; count: number } expect(second.count).toBe(first.count + 2) expect(second.hash).not.toBe(first.hash) - expect(store.takeStale()).toEqual([]) + expect(owed()).toMatchObject({ state: 'current', fail_count: 0 }) }) it('appends onto a file it read through and decoded no session from', async () => { @@ -76,7 +83,7 @@ it('appends onto a file it read through and decoded no session from', async () = outcome: { session: null, byteOffset: 100 } }) expect(cursor()).toBe(100) - expect(store.takeStale()).toEqual([]) + expect(owed()).toMatchObject({ state: 'current', fail_count: 0 }) replayTranscriptRead({ mode: 'append', @@ -87,7 +94,7 @@ it('appends onto a file it read through and decoded no session from', async () = expect(indexedMessages()).toBe(2) expect(cursor()).toBe(220) - expect(store.takeStale()).toEqual([]) + expect(owed()).toMatchObject({ state: 'current', fail_count: 0 }) }) it('declines an append that starts past its own cursor and records the file', async () => { @@ -107,7 +114,7 @@ it('declines an append that starts past its own cursor and records the file', as expect(indexedMessages()).toBe(3) expect(cursor()).toBe(100) - expect(store.takeStale().map((candidate) => candidate.file.path)).toEqual([SYNTHETIC_TRANSCRIPT]) + expect(owed()).toMatchObject({ state: 'due' }) }) it('declines a file whose identity changed under the same path', async () => { @@ -127,7 +134,7 @@ it('declines a file whose identity changed under the same path', async () => { }) expect(indexedMessages()).toBe(2) - expect(store.takeStale()).toHaveLength(1) + expect(owed()?.state).not.toBe('current') }) it('never advances the cursor for an incomplete read', async () => { @@ -152,7 +159,7 @@ it('never advances the cursor for an incomplete read', async () => { } ).n ).toBe(3) - expect(store.takeStale()).toHaveLength(1) + expect(owed()?.state).not.toBe('current') }) it('indexes nothing at all from a read that was incomplete from the start', async () => { @@ -167,7 +174,11 @@ it('indexes nothing at all from a read that was incomplete from the start', asyn expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 0 }) - expect(cursor()).toBeUndefined() + // No cursor, because nothing was read through. The row exists all the same: + // it is where the failure is counted, and a file that fails on its first read + // is exactly the one that has no row of its own to count on. + expect(cursor()).toBe(0) + expect(owed()).toMatchObject({ state: 'failed', fail_count: 1 }) }) it('drops a file whose parser returned no session', async () => { @@ -207,7 +218,9 @@ it('writes nothing for a source whose parser cannot reach the channel', async () expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ n: 0 }) - expect(store.takeStale()).toEqual([]) + // No row at all, which is the record: the next pass reads a path the + // file table does not name. + expect(owed()).toBeUndefined() }) it('ignores a candidate older than the retention cutoff', async () => { @@ -217,53 +230,9 @@ it('ignores a candidate older than the retention cutoff', async () => { expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ n: 0 }) - expect(store.takeStale()).toEqual([]) -}) - -it('stops writing while the store refuses writes, but remembers what it skipped', async () => { - store.setAcceptingWrites(false) - replayTranscriptRead({ messages: userMessages('paused', 3) }) - - expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ - n: 0 - }) - expect(errors).toEqual([]) - // A pause is exactly the window in which every read is declined. Forgetting - // them would leave the whole paused span unindexed with nothing to replay it. - expect(store.takeStale().map((candidate) => candidate.file.path)).toEqual([SYNTHETIC_TRANSCRIPT]) -}) - -it('keeps the paused re-read set when the retention window is reconfigured', async () => { - store.setAcceptingWrites(false) - replayTranscriptRead({ messages: userMessages('paused', 2) }) - expect(store.pendingFileCount).toBe(1) - - // The set records what still has to be read, not what is worth keeping. A - // window that now excludes this file is enforced where the re-read is - // dispatched, so nothing is written and the file leaves the set there. - store.setRetentionCutoffMs(Date.now()) - expect(store.pendingFileCount).toBe(1) - - store.setAcceptingWrites(true) - expect(store.takeStale()).toHaveLength(1) - replayTranscriptRead({ messages: userMessages('outside the window now', 2) }) - expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ - n: 0 - }) - expect(store.pendingFileCount).toBe(0) -}) - -it('drops the oldest record rather than growing without a bound, and says so', () => { - store.setAcceptingWrites(false) - for (let index = 0; index < STALE_PATH_LIMIT + 5; index++) { - store.markStale(syntheticCandidate({ path: `/transcript-${index}.jsonl` })) - } - - expect(store.pendingFileCount).toBe(STALE_PATH_LIMIT) - expect(store.droppedPendingFileCount).toBe(5) - const kept = store.takeStale().map((candidate) => candidate.file.path) - expect(kept).not.toContain('/transcript-0.jsonl') - expect(kept).toContain(`/transcript-${STALE_PATH_LIMIT + 4}.jsonl`) + // No row at all, which is the record: the next pass reads a path the + // file table does not name. + expect(owed()).toBeUndefined() }) it('keeps the session list running when the index write fails', async () => { @@ -282,7 +251,7 @@ it('keeps the session list running when the index write fails', async () => { }) ).not.toThrow() expect(errors.length).toBeGreaterThan(0) - expect(store.takeStale()).toHaveLength(1) + expect(owed()?.state).not.toBe('current') }) it('unregisters cleanly, leaving later reads unindexed', async () => { @@ -369,5 +338,5 @@ it('keeps a proven file identity when a later read cannot stat it', async () => expect(indexedMessages()).toBe(4) expect(cursor()).toBe(200) - expect(store.takeStale()).toHaveLength(1) + expect(owed()?.state).not.toBe('current') }) diff --git a/src/main/ai-vault-search/session-search-index-consumer.ts b/src/main/ai-vault-search/session-search-index-consumer.ts index e4fa6da4a8e..a457ca5e7c3 100644 --- a/src/main/ai-vault-search/session-search-index-consumer.ts +++ b/src/main/ai-vault-search/session-search-index-consumer.ts @@ -7,6 +7,7 @@ import { type TranscriptReadOutcome, type TranscriptReadStart } from '../ai-vault/session-transcript-consumers' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' import { fileIdentity } from './session-search-file-cursor' import type { SessionSearchFileWrite } from './session-search-index-writer' import type { SessionSearchStore } from './session-search-store' @@ -16,30 +17,22 @@ import type { SessionSearchStore } from './session-search-store' * * It keeps its own cursor in the `files` table and never consults the parse * cache: the two answer different questions and diverge the moment either - * declines a read. Three refusals, each of which leaves the cursor where it - * was and records the file for a later whole re-read: + * declines a read. * - * - `beginRead` returns null when this index's cursor is behind the offset an - * `append` continues from, or when the file's identity changed. - * - a buffering failure stops the read's rows without failing the session list. - * - an `incomplete` outcome never commits; those rows are not the whole span. + * Every refusal leaves the cursor where it was and writes what the next pass + * needs on the row itself, because the row is the only thing that outlives this + * read. A declined append is `due`: the index is behind on a span no append + * reaches, so the file has to be read whole. A read that started and did not + * commit is `failed`, counted, and stamped with the stat it failed at, which is + * what stops an unreadable transcript being retried on every pass for ever. */ export class SessionSearchIndexConsumer implements TranscriptConsumer { constructor(private readonly store: SessionSearchStore) {} beginRead(start: TranscriptReadStart): TranscriptReadConsumer | null { const { candidate } = start - if (!this.store.acceptsCandidate(candidate)) { - // A pause is a reason not to write now, not a reason to forget the read. - // `markStale` applies the retention rule itself, so a candidate that is - // out of scope rather than merely paused is still dropped here. - this.store.markStale(candidate) - return null - } - // A parser that decodes where the channel cannot reach it reports every read - // as incomplete. Declining here is not the same as being behind: no re-read - // would help, so the file is not recorded either. if (!parserPublishesMessages(candidate)) { + this.noteUnreachableParser(candidate) return null } if (start.mode === 'append') { @@ -48,7 +41,8 @@ export class SessionSearchIndexConsumer implements TranscriptConsumer { // This index never saw the span before `previousByteOffset`; appending // here would leave a hole no later read can fill. A null cursor is the // file a chunked read left half written, which no offset continues. - this.store.markStale(candidate) + // Either way the next pass has to read this file from the start. + this.store.setFileState(candidate.file.path, 'due') return null } } @@ -59,11 +53,42 @@ export class SessionSearchIndexConsumer implements TranscriptConsumer { start.identity ) if (!write) { - this.store.markStale(candidate) + // A closed store, a candidate outside the retention window, or a row that + // moved under this read. Only a row that exists has anything to record. + this.store.setFileState(candidate.file.path, 'due') return null } return new SessionSearchReadConsumer(this.store, start, write) } + + /** + * A source no read can ever index, recorded as one this index has seen. + * + * A parser that decodes where the message channel cannot reach it -- OpenCode's + * SQLite sessions today -- publishes nothing, so no read of it will ever + * commit a row. Leaving the file table silent about it is not free: the next + * pass sees a path the index holds nothing for, asks for a read, and asking + * over a warm cache drops the session list's own resume point. The sidebar's + * fold is thrown away and the whole database is decoded again, on every pass, + * for ever. + * + * The row written is the shape the store already has for a read that went + * through and decoded no session: cursor at the file's size, no session row. + * The decide step then skips it until its stat moves, and the retirement walk + * retires it like any other row when it goes. + */ + private noteUnreachableParser(candidate: SessionFileCandidate): void { + const write = this.store.beginWrite(candidate, 'replace', 0) + const committed = + write?.commit({ + session: null, + byteOffset: candidate.file.sizeBytes ?? 0, + incomplete: false + }) === true + if (committed) { + this.store.writeCommitted(candidate) + } + } } class SessionSearchReadConsumer implements TranscriptReadConsumer { @@ -105,7 +130,10 @@ class SessionSearchReadConsumer implements TranscriptReadConsumer { this.store.writeCommitted(candidate) return } - this.store.markStale(candidate) + // Counted against the stat it failed at, not merely recorded: a transcript + // the reader cannot open fails identically on every pass, and only a change + // to this stat can mean the file itself changed. + this.store.setFileState(candidate.file.path, 'failed', candidate.file.mtimeMs) } } diff --git a/src/main/ai-vault-search/session-search-index-pass.test.ts b/src/main/ai-vault-search/session-search-index-pass.test.ts new file mode 100644 index 00000000000..c32eec59a96 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-pass.test.ts @@ -0,0 +1,194 @@ +import { appendFile, rm, stat, utimes } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { runSessionSearchIndexPass } from './session-search-index-pass' +import { parseTranscript } from './session-search-transcript-fixtures' +import { + claudeLines, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' +import { discoverSessionSearchCandidates } from './session-search-scan-roots' +import { SessionSearchStore } from './session-search-store' + +const FIRST = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +const SECOND = 'bbbbbbbb-cccc-4ddd-8eee-ffffffffffff' + +let harness: SessionSearchIndexerHarness +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + harness = await openSessionSearchIndexerHarness('ss-index-pass') + await writeClaudeTranscript(transcript(FIRST), ['the first transcript'], FIRST) + await writeClaudeTranscript(transcript(SECOND), ['the second transcript'], SECOND) + store = openStore() +}) + +afterEach(async () => { + resetTranscriptConsumersForTests() + store.close() + await harness.cleanup() +}) + +function transcript(sessionId: string): string { + return join(harness.claudeProjectDir, `${sessionId}.jsonl`) +} + +function openStore(): SessionSearchStore { + const opened = new SessionSearchStore(harness.databasePath, (error) => errors.push(error)) + registerSessionSearchIndexConsumer(opened) + return opened +} + +async function candidates() { + return ( + await discoverSessionSearchCandidates(harness.roots, { + limitPerAgent: Number.POSITIVE_INFINITY + }) + ).candidates +} + +/** What a pass hands the read loop: the store's rows, read once. */ +function rows() { + return new Map(store.files().map((row) => [row.path, row])) +} + +function pass(options: { overdue?: () => boolean } = {}) { + return runSessionSearchIndexPass(store, [], { rows: rows(), ...options }) +} + +async function passOverAll(options: { overdue?: () => boolean } = {}) { + return runSessionSearchIndexPass(store, await candidates(), { rows: rows(), ...options }) +} + +function states(): Record { + return Object.fromEntries(store.files().map((row) => [row.path, row.state])) +} + +it('re-reads nothing it already holds, even with a cold session-list cache', async () => { + const first = await passOverAll() + expect(first.stats.fullParses).toBe(2) + + // A restart: the parse cache is gone, the index's `files` table is not. + store.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store = openStore() + + const second = await passOverAll() + expect(second.stats).toMatchObject({ fullParses: 0, incremental: 0, reused: 0, bytesRead: 0 }) + expect(errors).toEqual([]) +}) + +it('resumes into a grown transcript instead of re-reading it whole', async () => { + await passOverAll() + await appendFile(transcript(FIRST), `${claudeLines(['a later turn'], FIRST, 10).join('\n')}\n`) + + const second = await passOverAll() + expect(second.stats).toMatchObject({ incremental: 1, fullParses: 0 }) +}) + +// Nothing is recorded about what a deadline cut off, because being owed is a +// fact about the row: the file is read on the next pass for the same reason it +// was owed on this one. +it('leaves what it ran out of time for owed, with nothing written down', async () => { + const all = await candidates() + const cut = await runSessionSearchIndexPass(store, all, { rows: rows(), overdue: () => true }) + + expect(cut.outOfTime).toBe(true) + expect(store.files()).toHaveLength(1) + const second = await passOverAll() + expect(second.stats.fullParses).toBe(1) + expect(store.files()).toHaveLength(2) +}) + +// The deadline is never applied before the pass has read anything, so a single +// transcript larger than one deadline is read alone rather than starved. +it('reads one file even when the deadline has already expired', async () => { + const only = (await candidates()).slice(0, 1) + const alone = await runSessionSearchIndexPass(store, only, { rows: rows(), overdue: () => true }) + + expect(alone.outOfTime).toBe(false) + expect(store.files()).toHaveLength(1) +}) + +it('skips a source the reader cannot even open without failing the pass', async () => { + const all = await candidates() + await rm(transcript(FIRST)) + await runSessionSearchIndexPass(store, all, { rows: rows() }) + + // One session indexed, and the missing one recorded as a failed read rather + // than as content the index holds. + expect(harness.read((db) => db.prepare('SELECT count(*) AS n FROM sessions').get())).toEqual({ + n: 1 + }) + expect(states()[transcript(FIRST)]).toBe('failed') +}) + +// Finding 6: mtime alone is not the freshness key. A transcript that grows +// while keeping its mtime (a same-second append, a restored timestamp) is a +// different file to the index, and reading only mtime would skip it forever. +it('re-reads a file that grew without its mtime moving', async () => { + const path = transcript(FIRST) + // A whole-millisecond stamp, so restoring it later reproduces it exactly. + const frozen = new Date(1_740_000_000_000) + await utimes(path, frozen, frozen) + await passOverAll() + + await appendFile(path, `${claudeLines(['a same-mtime append'], FIRST, 20).join('\n')}\n`) + await utimes(path, frozen, frozen) + expect((await stat(path)).mtimeMs).toBe(frozen.getTime()) + + const second = await passOverAll() + expect(second.stats.fullParses + second.stats.incremental).toBe(1) +}) + +// Finding 5: the decision reads the session list's cache and then changes it, +// so outside the per-path lane an overlapping list parse stores its entry in +// between and the forced read degrades into a reuse. +it('is not overtaken by a list parse racing the same path', async () => { + const path = transcript(FIRST) + const all = await candidates() + const only = all.filter((candidate) => candidate.file.path === path) + + // The list parses this path first, so its cursor covers the file, and again + // concurrently with the index's pass so the two interleave. + await parseTranscript(path) + await Promise.all([ + parseTranscript(path), + runSessionSearchIndexPass(store, only, { rows: rows() }) + ]) + + expect(harness.read((db) => db.prepare('SELECT count(*) AS n FROM sessions').get())).toEqual({ + n: 1 + }) +}) + +// Finding 4d: a declined read is a parse that returns normally and indexes +// nothing. It has to leave the row owing a read, not looking covered. +it('leaves a declined read owed rather than recorded as held', async () => { + const only = (await candidates()).slice(0, 1) + // What a store that refuses a write looks like from the consumer's side: the + // read runs, and nothing is written. + store.beginWrite = () => null + + const stats = await runSessionSearchIndexPass(store, only, { rows: rows() }) + + expect(stats.stats.fullParses).toBe(1) + expect(harness.read((db) => db.prepare('SELECT count(*) AS n FROM sessions').get())).toEqual({ + n: 0 + }) + expect(store.files()).toEqual([]) +}) + +it('reads nothing when there is nothing to read', async () => { + expect((await pass()).stats).toMatchObject({ fullParses: 0 }) +}) diff --git a/src/main/ai-vault-search/session-search-index-pass.ts b/src/main/ai-vault-search/session-search-index-pass.ts new file mode 100644 index 00000000000..05ccea1a314 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-pass.ts @@ -0,0 +1,84 @@ +import { throwIfAiVaultScanCancelled } from '../ai-vault/ai-vault-scan-cancellation' +import { + createSessionParseStats, + parseAgentSessionFileCached, + type SessionParseStats +} from '../ai-vault/session-scanner-parse-cache' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import { fileIdentity } from './session-search-file-cursor' +import { sessionSearchReadDecision } from './session-search-read-decision' +import type { SessionSearchFileRow, SessionSearchStore } from './session-search-store' + +export type SessionSearchIndexPassOptions = { + signal?: AbortSignal + /** The store's rows for this pass, read once. Absent means the index holds nothing. */ + rows: ReadonlyMap + /** + * True once the pass has spent its wall-clock deadline. The one bound on how + * long a pass reads for: files and bytes are proxies for time, and the thing + * worth capping is the share of the wall clock an unasked background index + * takes. Never applied before the pass has read anything, so an oversized + * transcript is read alone rather than deferred for ever. + */ + overdue?: () => boolean +} + +/** + * Reads whatever the decide step says is owed, until the deadline. + * + * Nothing is recorded about what it did not reach. A candidate the deadline cut + * off is still owed on the next pass for the same reason it was owed on this + * one — its row says so — so there is no queue to keep, nothing to bound, and + * nothing to drop. What the reads themselves leave behind is written by the + * index consumer onto the rows. + */ +export async function runSessionSearchIndexPass( + store: SessionSearchStore, + candidates: readonly SessionFileCandidate[], + options: SessionSearchIndexPassOptions +): Promise<{ stats: SessionParseStats; outOfTime: boolean }> { + const stats = createSessionParseStats() + const cutoffMs = store.retentionCutoff + let read = 0 + let outOfTime = false + for (const candidate of candidates) { + throwIfAiVaultScanCancelled(options.signal) + const path = candidate.file.path + const row = options.rows.get(path) + const decision = sessionSearchReadDecision({ + candidate, + row, + // Only asked for a path the index holds something for; for the rest the + // decision is already made and this would be a query per new file. + cursor: row ? store.indexedFile(path, fileIdentity(candidate.file)) : null, + cutoffMs + }) + if (decision === 'skip') { + continue + } + // The decide step is one cursor lookup, so it runs for the whole list even + // once the deadline has gone: knowing what is owed costs nothing, and the + // count of what a pass left is worth more than the microseconds. + outOfTime ||= read > 0 && options.overdue?.() === true + if (outOfTime) { + continue + } + // The clock the deadline reads is one the owner may close behind: the read + // below writes to the store, so stop here rather than on a shut handle. + throwIfAiVaultScanCancelled(options.signal) + read += 1 + try { + await parseAgentSessionFileCached(candidate, process.platform, stats, decision) + } catch (error) { + throwIfAiVaultScanCancelled(options.signal) + // The reader reports a read it could not finish to the consumer, which is + // what records the failure on the row; nothing is counted here. + console.warn( + '[ai-vault-search] indexing skipped', + candidate.agent, + error instanceof Error ? error.name : 'ParseError' + ) + } + } + return { stats, outOfTime } +} diff --git a/src/main/ai-vault-search/session-search-index-writer.test.ts b/src/main/ai-vault-search/session-search-index-writer.test.ts index 46d147dcce0..1be12e35bc7 100644 --- a/src/main/ai-vault-search/session-search-index-writer.test.ts +++ b/src/main/ai-vault-search/session-search-index-writer.test.ts @@ -112,13 +112,12 @@ it('refuses to commit a write whose file was removed mid-read', () => { it('declines a behind cursor in beginRead before it ever reaches the store', () => { const attempted: number[] = [] const stub = { - acceptsCandidate: () => true, indexedFile: () => ({ byteOffset: 100, mtimeMs: 1, sizeBytes: 1 }), beginWrite: (_candidate: unknown, _mode: unknown, previousByteOffset: number) => { attempted.push(previousByteOffset) return { add: () => undefined, commit: () => true } }, - markStale: () => undefined + setFileState: () => undefined } as unknown as SessionSearchStore const consumer = new SessionSearchIndexConsumer(stub) @@ -144,7 +143,6 @@ it('declines a behind cursor in beginRead before it ever reaches the store', () it("hands the read's identity accessor to the store", () => { const captured: unknown[] = [] const stub = { - acceptsCandidate: () => true, indexedFile: () => null, beginWrite: ( _candidate: unknown, @@ -155,7 +153,7 @@ it("hands the read's identity accessor to the store", () => { captured.push(identity) return { add: () => undefined, commit: () => true } }, - markStale: () => undefined + setFileState: () => undefined } as unknown as SessionSearchStore const identity = (): null => null diff --git a/src/main/ai-vault-search/session-search-indexer-options.ts b/src/main/ai-vault-search/session-search-indexer-options.ts new file mode 100644 index 00000000000..08eb4aeb2af --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer-options.ts @@ -0,0 +1,49 @@ +import type { SessionSearchClock } from './session-search-clock' +import type { SessionSearchScanRoots } from './session-search-scan-roots' + +/** Default cycle. Long enough that a machine with thousands of transcripts is + * not re-statting continuously, short enough that a live conversation shows up + * while the user is still in it. */ +export const DEFAULT_SESSION_SEARCH_RECONCILE_INTERVAL_MS = 20_000 +/** Newest-N per agent root: the same recency rule the session sidebar applies. */ +export const DEFAULT_SESSION_SEARCH_RECENT_PER_AGENT = 12 +/** + * A quarter of the interval: the only bound on how long one pass reads for. + * + * The timer re-arms after a pass settles, so a pass that spends its whole + * deadline is followed by a full interval of quiet — five seconds of reading in + * every twenty-five, a fifth of the wall clock, and the stated ceiling is a + * quarter. Files the deadline cut off go back on the queue at full speed rather + * than being read slowly, which is what a load-average back-off did instead. + */ +export const DEFAULT_SESSION_SEARCH_PASS_DEADLINE_FRACTION = 4 +/** + * Cycles between whole-machine sweeps: five minutes at the default interval. + * + * A sweep is the only pass that sees a file nothing has told the indexer about + * — an old transcript deleted, a root that came back, a tree restored from a + * backup — so the cadence is what replaces every re-arm-on-recovery rule. A + * warm sweep is stats and readdirs, not reads, because the pass skips anything + * the index already covers at its current stat. + */ +export const DEFAULT_SESSION_SEARCH_FULL_SWEEP_EVERY_CYCLES = 15 + +/** + * Everything an indexer is. Immutable after construction: a settings change is + * `close()` and a new instance, which is also how the index is thrown away + * (`close()`, `removeSessionSearchDatabase(databasePath)`, construct again). + */ +export type SessionSearchIndexerOptions = { + databasePath: string + roots: SessionSearchScanRoots + /** null = all history; otherwise only transcripts modified within this many days. */ + historyDays: number | null + clock?: SessionSearchClock + reconcileIntervalMs?: number + recentPerAgent?: number + /** Wall time one pass may read for; the rest goes back on the queue. */ + passDeadlineMs?: number + /** Cycles between whole-machine sweeps. */ + fullSweepEveryCycles?: number + onError?: (error: unknown) => void +} diff --git a/src/main/ai-vault-search/session-search-indexer-test-fixture.ts b/src/main/ai-vault-search/session-search-indexer-test-fixture.ts new file mode 100644 index 00000000000..f8510510807 --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer-test-fixture.ts @@ -0,0 +1,168 @@ +import { mkdir, mkdtemp, rename, rm, stat, utimes, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import SyncDatabase from '../sqlite/sync-database' +import { isolatedScanRoots } from '../ai-vault/session-scanner-test-fixtures' +import type { SessionSearchClock, SessionSearchTimerHandle } from './session-search-clock' +import type { SessionSearchScanRoots } from './session-search-scan-roots' +import { assistantRecord, userRecord } from './session-search-transcript-fixtures' + +const CLOCK_EPOCH_MS = 1_740_000_000_000 + +/** Wall time the indexer's guarantee is stated in, under the test's control. */ +export class FakeSessionSearchClock implements SessionSearchClock { + private time = CLOCK_EPOCH_MS + private nextId = 1 + private nowCalls = 0 + private readonly timers = new Map void }>() + + /** + * What each `now()` reading costs. A pass reads the clock once per file it is + * about to read, so this is how a test spends a pass's deadline without + * waiting: it is the wall time the reads themselves take. + */ + costPerNowMs = 0 + + /** + * Runs on every `now()`, with the call number. The only synchronous seam into + * a running pass: the deadline check is what a pass consults between files. + */ + onNow: ((call: number) => void) | null = null + + now(): number { + const at = this.time + this.time += this.costPerNowMs + this.onNow?.(++this.nowCalls) + return at + } + + setTimeout(callback: () => void, ms: number): SessionSearchTimerHandle { + const id = this.nextId++ + this.timers.set(id, { at: this.time + ms, callback }) + return id + } + + clearTimeout(handle: SessionSearchTimerHandle): void { + this.timers.delete(handle as number) + } + + /** Moves time forward and fires every timer that came due, in order. */ + advance(ms: number): void { + this.time += ms + for (const [id, timer] of [...this.timers].sort((left, right) => left[1].at - right[1].at)) { + if (timer.at <= this.time) { + this.timers.delete(id) + timer.callback() + } + } + } + + get pendingTimers(): number { + return this.timers.size + } +} + +export type SessionSearchIndexerHarness = { + root: string + databasePath: string + roots: SessionSearchScanRoots + claudeProjectDir: string + /** A second connection: the store keeps its own private. */ + read: (query: (db: SyncDatabase) => T) => T + /** Plants what a killed writer would have left; nothing in the app writes here. */ + write: (query: (db: SyncDatabase) => T) => T + cleanup: () => Promise +} + +export async function openSessionSearchIndexerHarness( + name: string +): Promise { + const root = await mkdtemp(join(tmpdir(), `${name}-`)) + const roots = isolatedScanRoots(root) + const databasePath = join(root, 'index', 'index.sqlite') + return { + root, + databasePath, + roots, + claudeProjectDir: join(roots.claudeProjectsDir, 'project'), + read: (query) => withConnection(databasePath, true, query), + write: (query) => withConnection(databasePath, false, query), + cleanup: () => rm(root, { recursive: true, force: true }) + } +} + +function withConnection( + path: string, + readonlyConnection: boolean, + query: (db: SyncDatabase) => T +): T { + const db = new SyncDatabase(path, { readonly: readonlyConnection }) + try { + return query(db) + } finally { + db.close() + } +} + +/** A native-chat-shaped Claude transcript: the same records the app itself writes. */ +export async function writeClaudeTranscript( + path: string, + turns: readonly string[], + sessionId: string +): Promise { + await mkdir(dirname(path), { recursive: true }) + await writeFile(path, `${claudeLines(turns, sessionId, 0).join('\n')}\n`) +} + +export function claudeLines( + turns: readonly string[], + sessionId: string, + startIndex: number +): string[] { + return turns.flatMap((turn, offset) => [ + userRecord(startIndex + offset * 2, turn, sessionId), + assistantRecord(startIndex + offset * 2 + 1, `noted: ${turn}`, sessionId) + ]) +} + +/** + * Replaces a transcript the way an editor or a sync client does: a new inode + * renamed over the old name. Same byte length on purpose, so the only thing + * that can tell the two files apart is their filesystem identity. + */ +export async function renameReplaceTranscript( + path: string, + turns: readonly string[], + sessionId: string +): Promise { + const before = await stat(path) + const replacement = `${path}.replacement` + await writeClaudeTranscript(replacement, turns, sessionId) + await rename(replacement, path) + const later = new Date(before.mtimeMs + 5_000) + await utimes(path, later, later) +} + +/** + * A message-graph transcript, the shape OpenClaw, Pi, OMP and Prime Agent + * write. The session id comes from the file name, so callers name the file. + */ +export async function writeMessageGraphTranscript( + path: string, + turns: readonly string[] +): Promise { + await mkdir(dirname(path), { recursive: true }) + const lines = turns.flatMap((turn, index) => [ + JSON.stringify({ + type: 'message', + timestamp: new Date(CLOCK_EPOCH_MS + index * 120_000).toISOString(), + message: { role: 'user', content: turn } + }), + JSON.stringify({ + type: 'message', + timestamp: new Date(CLOCK_EPOCH_MS + index * 120_000 + 60_000).toISOString(), + message: { role: 'assistant', content: `noted: ${turn}` } + }) + ]) + await writeFile(path, `${lines.join('\n')}\n`) +} diff --git a/src/main/ai-vault-search/session-search-indexer.test.ts b/src/main/ai-vault-search/session-search-indexer.test.ts new file mode 100644 index 00000000000..5985628ba2d --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer.test.ts @@ -0,0 +1,1090 @@ +import { existsSync, mkdirSync, rmSync, utimesSync, writeFileSync } from 'node:fs' +import { appendFile, chmod, mkdir, rm, stat, utimes } from 'node:fs/promises' +import { dirname, join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { removeSessionSearchDatabase } from './session-search-schema' +import { parseTranscript } from './session-search-transcript-fixtures' +import { + claudeLines, + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + renameReplaceTranscript, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +const INTERVAL_MS = 20_000 +// chmod cannot deny root, and Windows ignores the mode bits entirely, so the +// two refusal tests would assert on an unreached branch there. +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const SESSION_ID = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +const OTHER_SESSION_ID = 'bbbbbbbb-cccc-4ddd-8eee-ffffffffffff' +const SETTLED_SESSION_ID = 'dddddddd-cccc-4ddd-8eee-ffffffffffff' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-indexer') + indexer = null +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function newIndexer( + overrides: Partial[0]> = {} +) { + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS, + onError: (error) => errors.push(error), + ...overrides + }) + return indexer +} + +/** Sessions a published-view read returns for one term, the only legal shape. */ +function sessionsMatching(term: string): string[] { + return harness.read((db: SyncDatabase) => + ( + db + .prepare( + `SELECT DISTINCT s.session_id AS id FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY s.session_id` + ) + .all(term) as { id: string }[] + ).map((row) => row.id) + ) +} + +function indexedSessionCount(): number { + return harness.read( + (db: SyncDatabase) => + (db.prepare('SELECT count(*) AS n FROM sessions').get() as { n: number }).n + ) +} + +/** The row the store holds for a path, which is the indexer's whole memory of it. */ +function rowFor(path: string) { + return harness.read((db: SyncDatabase) => + db.prepare('SELECT state, fail_count AS failCount FROM files WHERE path = ?').get(path) + ) as { state: string; failCount: number } | undefined +} + +function fileState(path: string): string | undefined { + return rowFor(path)?.state +} + +/** The byte offset the index recorded; PR 2 stores -1 for a half-written file. */ +function indexedByteOffset(path: string): number | undefined { + return harness.read( + (db: SyncDatabase) => + ( + db.prepare('SELECT byte_offset AS offset FROM files WHERE path = ?').get(path) as + | { offset: number } + | undefined + )?.offset + ) +} + +/** What a chunk of a read that never finished leaves on the file row. */ +function plantPartialCursor(path: string): void { + harness.write((db: SyncDatabase) => + db.prepare('UPDATE files SET byte_offset = -1 WHERE path = ?').run(path) + ) +} + +function indexedCursor(path: string): { mtime_ms: number; size_bytes: number } | undefined { + return harness.read( + (db: SyncDatabase) => + db.prepare('SELECT mtime_ms, size_bytes FROM files WHERE path = ?').get(path) as + | { mtime_ms: number; size_bytes: number } + | undefined + ) +} + +function transcriptPath(name = SESSION_ID): string { + return join(harness.claudeProjectDir, `${name}.jsonl`) +} + +/** + * Starts the indexer over a root that already holds one indexed transcript, so + * the opening sweep is behind us and `reconcile()` runs a cycle. It is dated + * ahead of everything the caller writes afterwards, so it stays inside any + * recency window and is skipped rather than read. + */ +async function startAfterASweep( + overrides: Partial[0]> = {} +): Promise { + const settled = transcriptPath(SETTLED_SESSION_ID) + await writeClaudeTranscript(settled, ['a conversation from before'], SETTLED_SESSION_ID) + // Wall time, not the fake clock: recency is decided by real file mtimes. + const ahead = new Date(Date.now() + 3_600_000) + await utimes(settled, ahead, ahead) + await newIndexer(overrides).start() +} + +/** + * Makes every pass stop after `files` reads: the pass consults the clock once + * per file it is about to read, and each reading costs a quarter of the + * deadline it is measured against. + */ +function readsPerPass(files: number): { passDeadlineMs: number } { + clock.costPerNowMs = 1_000 + return { passDeadlineMs: files * 1_000 } +} + +/** Advances one reconcile interval and waits for the cycle it fires. */ +async function nextCycle(): Promise { + clock.advance(INTERVAL_MS) + await indexer?.settled() +} + +it('reflects a grown transcript within one reconcile interval', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['find the flaky terminal reattach'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('reattach')).toEqual([SESSION_ID]) + expect(sessionsMatching('quarantine')).toEqual([]) + + await appendFile( + path, + `${claudeLines(['quarantine the leaking pty'], SESSION_ID, 10).join('\n')}\n` + ) + await nextCycle() + + expect(sessionsMatching('quarantine')).toEqual([SESSION_ID]) + expect(errors).toEqual([]) +}) + +it('reflects a rename-replaced transcript within one reconcile interval', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['original content aaaa'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('original')).toEqual([SESSION_ID]) + const original = await stat(path) + + await renameReplaceTranscript(path, ['swapped content bbbbb'], SESSION_ID) + // Same length, different inode: only the identity check can tell them apart. + expect((await stat(path)).size).toBe(original.size) + await nextCycle() + + expect(sessionsMatching('swapped')).toEqual([SESSION_ID]) + expect(sessionsMatching('original')).toEqual([]) + expect(errors).toEqual([]) +}) + +it('retires a deleted transcript within one reconcile interval', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['a session about to be deleted'], SESSION_ID) + await writeClaudeTranscript( + transcriptPath(OTHER_SESSION_ID), + ['a surviving session'], + OTHER_SESSION_ID + ) + await newIndexer().start() + await nextCycle() + expect(sessionsMatching('deleted')).toEqual([SESSION_ID]) + + await rm(path) + await nextCycle() + + expect(sessionsMatching('deleted')).toEqual([]) + expect(sessionsMatching('surviving')).toEqual([OTHER_SESSION_ID]) +}) + +it.skipIf(!CAN_DENY_READ)( + 'keeps rows for a source it cannot stat, because loss of contact is not deletion', + async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['an unverifiable session'], SESSION_ID) + await newIndexer().start() + await nextCycle() + + // The tree is gone from discovery's point of view, but the transcript itself + // was never proven absent: an unreadable parent is not a deleted file. + await chmod(harness.claudeProjectDir, 0o000) + try { + await nextCycle() + expect(sessionsMatching('unverifiable')).toEqual([SESSION_ID]) + } finally { + await chmod(harness.claudeProjectDir, 0o755) + } + } +) + +it('resumes after close and reopen without re-reading what it already indexed', async () => { + await writeClaudeTranscript(transcriptPath(), ['first indexed session'], SESSION_ID) + await writeClaudeTranscript( + transcriptPath(OTHER_SESSION_ID), + ['second indexed session'], + OTHER_SESSION_ID + ) + await newIndexer().start() + const indexedRows = harness.read((db: SyncDatabase) => + db.prepare('SELECT count(*) AS n FROM messages').get() + ) + indexer?.close() + + // A restart is a cold parse cache over a warm index; only the `files` table + // can say what has already been read. + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + const reopened = newIndexer() + await reopened.start() + + // `filesIndexed` is the count of rows the index holds at their current stat, + // so it stays 2. That nothing was opened again is the read loop's own test. + expect(reopened.status()).toMatchObject({ filesIndexed: 2, filesDue: 0 }) + expect( + harness.read((db: SyncDatabase) => db.prepare('SELECT count(*) AS n FROM messages').get()) + ).toEqual(indexedRows) + expect(sessionsMatching('indexed')).toEqual([SESSION_ID, OTHER_SESSION_ID].sort()) +}) + +// F12, as the immutable design states it: the history window is a construction +// argument, so widening it is a new instance whose opening sweep admits the +// older files, and narrowing it is the purge that opens every full sweep. +it('widens history by constructing a new instance and narrows by purging on its first sweep', async () => { + const fresh = transcriptPath() + const old = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(fresh, ['a recent conversation'], SESSION_ID) + await writeClaudeTranscript(old, ['an ancient conversation'], OTHER_SESSION_ID) + const longAgo = new Date(clock.now() - 120 * 86_400_000) + await utimes(old, longAgo, longAgo) + + // Newest-one per root, so the widened-in transcript is outside the recency + // window a cycle re-stats: only a full sweep can reach it. + await newIndexer({ historyDays: 30, recentPerAgent: 1 }).start() + expect(sessionsMatching('recent')).toEqual([SESSION_ID]) + expect(sessionsMatching('ancient')).toEqual([]) + + // Widening cannot be served from the index: those files were never read. + indexer?.close() + await newIndexer({ historyDays: null, recentPerAgent: 1 }).start() + expect(sessionsMatching('ancient')).toEqual([OTHER_SESSION_ID]) + + indexer?.close() + await newIndexer({ historyDays: 30, recentPerAgent: 1 }).start() + expect(sessionsMatching('ancient')).toEqual([]) + expect(sessionsMatching('recent')).toEqual([SESSION_ID]) +}) + +it.skipIf(!CAN_DENY_READ)( + 'names an unreadable root as degraded and keeps indexing the others', + async () => { + const blocked = join(harness.roots.codexSessionsDir ?? '', 'blocked') + await mkdir(blocked, { recursive: true }) + await writeClaudeTranscript(transcriptPath(), ['a readable claude session'], SESSION_ID) + await chmod(harness.roots.codexSessionsDir ?? '', 0o000) + try { + await newIndexer().start() + const status = indexer?.status() + expect(status?.phase).toBe('degraded') + expect(status?.degradedRoots.map((root) => root.root)).toContain( + harness.roots.codexSessionsDir + ) + expect(status?.degradedRoots[0]?.reason).toBeTruthy() + // A degraded root is not a degraded index: everything else still lands. + expect(sessionsMatching('readable')).toEqual([SESSION_ID]) + } finally { + await chmod(harness.roots.codexSessionsDir ?? '', 0o755) + } + } +) + +// The one bound on a pass. What it does not reach is owed on the next pass for +// the same reason it was owed on this one -- its row says so, or it has no row +// -- so nothing is written down and nothing can be lost. +it('reads what one pass has time for and finishes the rest on the next', async () => { + // The sweep is behind us, so this is the reconciler fitting four new files + // into a deadline that stops it after two. + await startAfterASweep(readsPerPass(2)) + for (let index = 0; index < 4; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript( + transcriptPath(session), + [`deadlined session number ${index}`], + session + ) + } + await indexer?.reconcile() + expect(indexer?.status()).toMatchObject({ filesIndexed: 3, filesDue: 0 }) + + await indexer?.reconcile() + expect(sessionsMatching('deadlined')).toHaveLength(4) + expect(indexer?.status().filesIndexed).toBe(5) + + // Settled, and it stays settled: nothing changed, so the cycle after this + // one opens none of them. + clock.costPerNowMs = 0 + await nextCycle() + expect(indexer?.status()).toMatchObject({ filesIndexed: 5, phase: 'current' }) +}) + +// First enablement inside a running app is the normal case, not an edge: the +// session list has been scanning since launch, so every transcript already has +// a cursor sitting at its current stat and the index has nothing at all. +it('fills an empty index over a warm session-list cache on the first reconcile', async () => { + await startAfterASweep() + const path = transcriptPath() + await writeClaudeTranscript(path, ['scanned before the index existed'], SESSION_ID) + // An ordinary parse now reuses its cached fold and opens no file, so no + // consumer is asked and there is nothing for a decline to record. + await parseTranscript(path) + + await indexer?.reconcile() + + expect(sessionsMatching('scanned')).toEqual([SESSION_ID]) +}) + +it('fills an empty index over a warm session-list cache on the first sweep', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['scanned before the index existed'], SESSION_ID) + await parseTranscript(path) + + await newIndexer().start() + + expect(sessionsMatching('scanned')).toEqual([SESSION_ID]) +}) + +// Finding 1: a sweep cut short used to be abandoned part way through. A pass +// that hands reads back is not an unfinished sweep -- its discovery and its +// retirement both completed -- so it must not re-arm one, and the queue is what +// carries the reads it did not reach until the whole machine is covered. +it('covers the whole machine over the passes that follow a truncated sweep', async () => { + const sessions = Array.from( + { length: 20 }, + (_unused, index) => `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + ) + for (const session of sessions) { + await writeClaudeTranscript(transcriptPath(session), [`sweepwide session ${session}`], session) + } + + // One transcript a pass, so the opening sweep reaches a twentieth of them. + await newIndexer(readsPerPass(1)).start() + expect(indexedSessionCount()).toBeGreaterThan(0) + expect(indexedSessionCount()).toBeLessThan(sessions.length) + + for (let cycle = 0; cycle < sessions.length; cycle++) { + await nextCycle() + } + + expect(indexedSessionCount()).toBe(sessions.length) + expect(indexer?.status().phase).toBe('current') +}) + +// Finding 2: the store's cutoff was set once at construction while purges used +// a fresh one, so a sweep deleted the row and the accept check re-indexed it. +it('moves the retention window with the clock instead of freezing it at construction', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['an entry that ages out'], SESSION_ID) + // Dated on the same clock the retention window is measured against. + const now = new Date(clock.now()) + await utimes(path, now, now) + await newIndexer({ historyDays: 1 }).start() + expect(sessionsMatching('ages')).toEqual([SESSION_ID]) + + clock.advance(3 * 86_400_000) + await indexer?.reconcile({ full: true }) + + expect(sessionsMatching('ages')).toEqual([]) + await nextCycle() + expect(sessionsMatching('ages')).toEqual([]) +}) + +// Round 10, H1. A cycle proves a deletion by comparing what the previous pass +// watched against what it discovers. A sweep used to watch only what it could +// not settle, which is nothing on a healthy machine, so the cycle after a sweep +// had no candidates at all and the cycle after that no longer remembered the +// file: a transcript deleted in that interval survived until the next sweep, +// up to `fullSweepEveryCycles` later. +it('retires a transcript deleted between a sweep and the cycle after it', async () => { + const going = transcriptPath() + const staying = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(going, ['a session deleted right after the sweep'], SESSION_ID) + await writeClaudeTranscript(staying, ['a surviving session'], OTHER_SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('deleted')).toEqual([SESSION_ID]) + + // No cycle in between: the sweep is the only pass that has seen this file. + await rm(going) + await nextCycle() + + expect(sessionsMatching('deleted')).toEqual([]) + expect(sessionsMatching('surviving')).toEqual([OTHER_SESSION_ID]) +}) + +// Round 10, M2. A sweep that throws part way learned nothing, and the flag that +// says one is owed was taken on entry. Losing it there leaves nothing armed to +// try again, so the machine outside the recency window goes unread until +// something else happens to ask for a sweep. +it('keeps a sweep due when the one that was running threw', async () => { + const older = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(older, ['an older conversation'], OTHER_SESSION_ID) + const yesterday = new Date(Date.now() - 86_400_000) + await utimes(older, yesterday, yesterday) + await writeClaudeTranscript(transcriptPath(), ['the newest conversation'], SESSION_ID) + + // Newest-one per root, so only a sweep can reach the older file. The clock is + // read inside the pass, which is where a failure part way through lands. + newIndexer({ recentPerAgent: 1 }) + let thrown = false + clock.onNow = () => { + if (thrown || indexedSessionCount() === 0) { + return + } + thrown = true + throw new Error('the sweep fell over') + } + await indexer?.start() + await indexer?.settled() + clock.onNow = null + + expect(errors.map((error) => (error as Error).message)).toEqual(['the sweep fell over']) + expect(sessionsMatching('older')).toEqual([]) + + // The pass after it is a sweep, not a cycle: a cycle reads one file per root. + await nextCycle() + expect(sessionsMatching('older')).toEqual([OTHER_SESSION_ID]) +}) + +// Round 10, M1. A transcript the reader cannot open is recorded stale by the +// consumer on every attempt, so it was re-read every cycle for ever: pending +// stuck at one, a failure count climbing without bound, and a phase that never +// left `indexing`. One file with the wrong mode bits read as a real backlog. +it.skipIf(!CAN_DENY_READ)('stops re-reading a transcript it cannot read', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['a session behind the wrong mode bits'], SESSION_ID) + await chmod(path, 0o000) + try { + await newIndexer().start() + for (let cycle = 0; cycle < 4; cycle++) { + await nextCycle() + } + + // Held out by its own row: three failures at one unchanged stat, counted on + // the row itself, and a phase that says the index knows it is not covering + // something rather than one that describes work it will never do. + expect(indexer?.status()).toMatchObject({ filesDue: 0, filesFailed: 1, phase: 'degraded' }) + expect(rowFor(path)?.failCount).toBeGreaterThanOrEqual(3) + + // And the hold is released by the only thing that can mean the file + // changed: its stat. + await chmod(path, 0o644) + const later = new Date(Date.now() + 60_000) + await utimes(path, later, later) + await nextCycle() + + expect(sessionsMatching('mode')).toEqual([SESSION_ID]) + expect(indexer?.status()).toMatchObject({ filesFailed: 0, phase: 'current' }) + } finally { + await chmod(path, 0o644) + } +}) + +// Round 10, M2. `close()` mid-pass left the pass reading a shut handle: three +// `database is not open` errors reached the owner, for a close they asked for. +it('reports nothing to its owner when it is closed part way through a pass', async () => { + await writeClaudeTranscript(transcriptPath(), ['one'], SESSION_ID) + const other = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(other, ['two'], OTHER_SESSION_ID) + const later = new Date(Date.now() + 60_000) + await utimes(other, later, later) + newIndexer() + + // Between two files: the pass reads the clock once per file it is about to + // read, and closing there is what a quit during a sweep looks like. + let closed = false + clock.onNow = () => { + if (closed || indexedSessionCount() === 0) { + return + } + closed = true + indexer?.close() + } + await indexer?.start() + await indexer?.settled() + clock.onNow = null + + expect(errors).toEqual([]) +}) + +// Round 10, M2, the other half: `status()` on a closed indexer opened a shut +// database, reported the failure, and answered zero files. +it('reports what it last knew after it is closed, without reading the database', async () => { + await writeClaudeTranscript(transcriptPath(), ['indexed before the close'], SESSION_ID) + await newIndexer().start() + expect(indexer?.status().filesIndexed).toBe(1) + + indexer?.close() + + expect(indexer?.status()).toMatchObject({ phase: 'closed', filesIndexed: 1 }) + expect(errors).toEqual([]) +}) + +// Round 10, L1. Two indexers on one database both register with the reader, so +// every transcript is read and written twice and the second write is fenced by +// the first at random. The recipe for every configuration change is +// close-then-construct, so the ordering that causes this is the one the recipe +// rules out; this is what says so rather than letting it corrupt quietly. +// Round 12, F2. The claim was staked before the store opened, so an open that +// threw left the path owned by an object that does not exist and every later +// construction was refused -- including the one that fixes whatever broke it. +it('releases the database path when the open itself throws', () => { + // A directory where the database file goes: the open fails, nothing is owned. + mkdirSync(harness.databasePath, { recursive: true }) + expect(() => newIndexer()).toThrow() + + rmSync(harness.databasePath, { recursive: true, force: true }) + expect(() => newIndexer()).not.toThrow() +}) + +it('refuses a second indexer on a database one already owns', () => { + newIndexer() + expect(() => newIndexer()).toThrow(/already has a live indexer/) +}) + +// PR 2 records a cursor no append continues for a file a chunked read left half +// written, and reports it as a null offset. The mtime and size on that row are +// the whole file's, so a freshness check comparing only those calls a prefix +// current and leaves it in the index for good. +it('re-reads a file a chunked read left half written, and settles it in one pass', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['the committed half'], SESSION_ID) + await newIndexer().start() + const whole = (await stat(path)).size + expect(indexedByteOffset(path)).toBe(whole) + indexer?.close() + + plantPartialCursor(path) + await newIndexer().start() + + // Nothing about the file changed, and it was read anyway: the whole of it, + // because there is no cursor to continue from. + expect(indexedByteOffset(path)).toBe(whole) + expect(fileState(path)).toBe('current') + expect(indexer?.status().phase).toBe('current') + indexer?.close() + + // A half-written file that also grew is repaired by one pass rather than two. + // The session list's resume point would have the reader offer an append here, + // and an append onto a partial cursor is a read the consumer declines. + plantPartialCursor(path) + await appendFile(path, `${claudeLines(['the lost half'], SESSION_ID, 10).join('\n')}\n`) + await newIndexer().start() + + expect(sessionsMatching('lost')).toEqual([SESSION_ID]) + expect(indexer?.status()).toMatchObject({ filesDue: 0, phase: 'current' }) +}) + +it('reports closed once it is closed, whatever it was doing before', async () => { + await writeClaudeTranscript(transcriptPath(), ['before the close'], SESSION_ID) + await newIndexer().start() + expect(indexer?.status().phase).toBe('current') + indexer?.close() + expect(indexer?.status().phase).toBe('closed') +}) + +// Finding 6: a queued entry carries the stat it was recorded with. Reading at +// that stat writes a cursor describing a file that no longer looks like this, +// so the next cycle distrusts it and re-reads it, forever. +it('reads a deferred file at its current stat, not the one the pass first saw', async () => { + // One file a pass, so the older one is left for the pass after this. + await startAfterASweep(readsPerPass(1)) + const older = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(older, ['the deferred conversation'], OTHER_SESSION_ID) + await writeClaudeTranscript(transcriptPath(), ['the newer conversation'], SESSION_ID) + const ahead = new Date((await stat(transcriptPath())).mtimeMs + 60_000) + await utimes(transcriptPath(), ahead, ahead) + + await indexer?.reconcile() + // No row for it at all, which is exactly why the next pass reads it. + expect(rowFor(older)).toBeUndefined() + + await appendFile( + older, + `${claudeLines(['appended while deferred'], OTHER_SESSION_ID, 10).join('\n')}\n` + ) + await indexer?.reconcile() + + expect(sessionsMatching('appended')).toEqual([OTHER_SESSION_ID]) + // The cursor has to describe the file as it is now; recorded against the + // stat the earlier pass saw it would be re-read on every cycle from here on. + const cursor = indexedCursor(older) + const current = await stat(older) + expect(cursor).toEqual({ mtime_ms: current.mtimeMs, size_bytes: current.size }) +}) + +// A declined read records the stat it was declined at. By the time the store +// hands it back the file has usually moved on again, and reading at the +// recorded stat writes a cursor the next cycle immediately distrusts. +it('reads a declined file at its current stat, not the one it was recorded with', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['the recorded conversation'], SESSION_ID) + await newIndexer().start() + + // A warm session-list cache over an empty index: the reader offers an append + // continuing an offset this index has never seen, so the consumer declines it + // and records the stat it declined at. + indexer?.close() + removeSessionSearchDatabase(harness.databasePath) + newIndexer() + await appendFile(path, `${claudeLines(['declined turn'], SESSION_ID, 10).join('\n')}\n`) + await parseTranscript(path) + // The index holds nothing for it, which is the record: a path the file table + // does not name is read from the start by the next pass. + expect(indexer?.status().filesIndexed).toBe(0) + + await appendFile(path, `${claudeLines(['later turn'], SESSION_ID, 20).join('\n')}\n`) + // The sweep is declined too -- the list's cursor is still ahead of the index + // -- so it is the pass after it that reads the file whole. + await indexer?.start() + await nextCycle() + + expect(sessionsMatching('later')).toEqual([SESSION_ID]) + const current = await stat(path) + expect(indexedCursor(path)).toEqual({ mtime_ms: current.mtimeMs, size_bytes: current.size }) +}) + +// Round 2, item 1: the sweep kept the rows and a cycle twenty seconds later +// deleted them, because the degraded-root fence was on the sweep path only. +it.skipIf(!CAN_DENY_READ)( + 'keeps an unlistable root through the cycles that follow the sweep', + async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + + await chmod(harness.roots.claudeProjectsDir ?? '', 0o000) + try { + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + + await nextCycle() + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + expect(indexer?.status().phase).toBe('degraded') + } finally { + await chmod(harness.roots.claudeProjectsDir ?? '', 0o755) + } + } +) + +// A root that cannot be listed is never believed to be empty, however many +// times it is asked: an error is not a listing, and only a listing is proof. +it.skipIf(!CAN_DENY_READ)('keeps an unlistable root degraded across repeated sweeps', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await newIndexer().start() + + await chmod(harness.roots.claudeProjectsDir ?? '', 0o000) + try { + for (let sweep = 0; sweep < 5; sweep++) { + await indexer?.reconcile({ full: true }) + } + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + expect(indexer?.status().phase).toBe('degraded') + } finally { + await chmod(harness.roots.claudeProjectsDir ?? '', 0o755) + } +}) + +// The first sweep of every process is exactly when a volume is most likely to +// be detached, and it is the pass with nothing behind it to compare against. +it('keeps a root that is gone at the first sweep after a restart', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await newIndexer().start() + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + // The volume is not there when the process comes back. + await rm(harness.roots.claudeProjectsDir ?? '', { recursive: true, force: true }) + await newIndexer().start() + + const status = indexer?.status() + expect(status?.phase).toBe('degraded') + expect(status?.degradedRoots.map((root) => root.root)).toContain(harness.roots.claudeProjectsDir) + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + + // And it clears once the volume is back. + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await indexer?.reconcile({ full: true }) + expect(indexer?.status()).toMatchObject({ phase: 'current', degradedRoots: [] }) +}) + +// Round 7: what the stateless walk costs, stated rather than hidden. A volume +// mounted at EXACTLY a configured root, unmounted so the mountpoint stays +// present and lists empty, is indistinguishable from a root the user emptied: +// there is no directory left whose absence could stop the walk. Inside one +// process the transition buys a pass of grace; across a restart there is no +// transition to see and the rows retire. The unmounts that actually happen are +// above the root, and the next test is the one that covers them. +it('retires an emptied configured root, one pass after it emptied', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on the mounted volume'], SESSION_ID) + await newIndexer().start() + + // The transcripts go; the root itself stays there and stays readable. + await rm(harness.claudeProjectDir, { recursive: true, force: true }) + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('mounted')).toEqual([SESSION_ID]) + expect(indexer?.status().phase).toBe('degraded') + + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('mounted')).toEqual([]) + expect(indexer?.status()).toMatchObject({ phase: 'current', degradedRoots: [] }) +}) + +// The same root, with no previous pass to compare against: nothing carries the +// transition across a restart, and the empty listing is proof on its own. +it('retires an emptied configured root at once on the first pass of a process', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on the mounted volume'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('mounted')).toEqual([SESSION_ID]) + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + await rm(harness.claudeProjectDir, { recursive: true, force: true }) + await newIndexer().start() + expect(sessionsMatching('mounted')).toEqual([]) +}) + +// The shape a real unmount takes: on Linux, WSL and sshfs the mountpoint is +// above the agent's root, so the root itself is missing. The walk stops at the +// root boundary and never asks the empty parent anything, which is what makes +// this hold with no memory on the first pass of a process. +it('proves nothing from an empty directory above the configured root', async () => { + for (let index = 0; index < 3; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`mounted session ${index}`], session) + } + await newIndexer().start() + expect(sessionsMatching('mounted')).toHaveLength(3) + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + // The volume that carried the agent's root is gone; what it was mounted + // under is still there, still listable, and empty of it. + await rm(harness.roots.claudeProjectsDir ?? '', { recursive: true, force: true }) + await newIndexer().start() + + const status = indexer?.status() + expect(status?.phase).toBe('degraded') + expect(status?.degradedRoots.map((root) => root.root)).toContain(harness.roots.claudeProjectsDir) + expect(sessionsMatching('mounted')).toHaveLength(3) +}) + +// A cycle only reads the newest N per agent, so a remounted volume would give +// up its newest transcript and keep the rest unreachable. Nothing watches for a +// recovery any more: the sweep cadence is what reaches it. +it('reads a root that came back on the next periodic sweep', async () => { + // Detached before anything was ever indexed, so the sweep correctly finds + // nothing and reports no alarm. + await newIndexer({ recentPerAgent: 1, fullSweepEveryCycles: 2 }).start() + expect(indexer?.status()).toMatchObject({ degradedRoots: [], filesIndexed: 0 }) + + for (let index = 0; index < 3; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`remounted session ${index}`], session) + } + + // Two cycles reach the newest one each; the sweep they are counting down to + // reads the rest. + await nextCycle() + await nextCycle() + expect(sessionsMatching('remounted')).toHaveLength(1) + + await nextCycle() + expect(sessionsMatching('remounted')).toHaveLength(3) +}) + +// Round 4, item 3: rows under no configured root. The walk judges each row on +// its own directory and proves nothing about one it cannot reach, so a profile +// that moved keeps its history rather than losing it. +it('keeps rows under no configured root, and retires them only when gone', async () => { + const moved = transcriptPath() + await writeClaudeTranscript(moved, ['a session in the old profile'], SESSION_ID) + await newIndexer().start() + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + // The profile moves: same index, a root that no longer covers those rows. + const elsewhere = join(harness.root, 'moved-profile') + newIndexer({ roots: { ...harness.roots, claudeProjectsDir: elsewhere } }) + await indexer?.start() + // Still on disk, so the rows stay: this is a configuration problem, not a + // licence to delete a user's history. + expect(sessionsMatching('profile')).toEqual([SESSION_ID]) + + await rm(moved) + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('profile')).toEqual([]) +}) + +// Round 7 replaced "only a census may conclude" with "whoever can prove it". +// A cycle walks the same directories and reaches the same verdict, so a project +// directory the user deleted does not wait for the next sweep. +it('lets a cycle retire a project directory the user deleted', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session about to vanish'], SESSION_ID) + await newIndexer().start() + + await rm(harness.claudeProjectDir, { recursive: true, force: true }) + // The pass that sees the root go from holding transcripts to holding none + // gives it one pass of grace, whether it is a sweep or a cycle. + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('vanish')).toEqual([SESSION_ID]) + + await nextCycle() + expect(sessionsMatching('vanish')).toEqual([]) + expect(indexer?.status()).toMatchObject({ phase: 'current', degradedRoots: [] }) +}) + +// C1: `close()` disarmed the timer and aborted the task in flight, but left the +// queue running, so a task queued a moment earlier still reopened a store and +// registered a consumer behind an indexer whose caller had finished with it. +it('stops everything on close, including work already queued', async () => { + await writeClaudeTranscript(transcriptPath(), ['indexed before the close'], SESSION_ID) + await newIndexer().start() + + const queued = indexer?.reconcile({ full: true }) + indexer?.close() + await queued + + // The queued pass never ran: had it run, it would have reached for a store + // this close had already shut, and reported the failure. + expect(errors).toEqual([]) + // And the timer is gone with it, so no later tick can queue another. + expect(clock.pendingTimers).toBe(0) + clock.advance(5 * INTERVAL_MS) + await indexer?.settled() + expect(errors).toEqual([]) + + // No store and no consumer: a scan after the close writes nothing. + const after = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(after, ['written after the close'], OTHER_SESSION_ID) + await parseTranscript(after) + expect(sessionsMatching('written')).toEqual([]) + expect(sessionsMatching('indexed')).toEqual([SESSION_ID]) +}) + +// What replaced `clear()`, exactly as the PR body documents it. The recipe is +// three statements because the indexer owns one store for one lifetime; the +// method it replaces owned a second one and had to keep the two in step. +it('throws the index away and rebuilds it by constructing a new instance', async () => { + await writeClaudeTranscript(transcriptPath(), ['indexed before the clear'], SESSION_ID) + await newIndexer().start() + expect(existsSync(harness.databasePath)).toBe(true) + + indexer?.close() + removeSessionSearchDatabase(harness.databasePath) + expect(existsSync(harness.databasePath)).toBe(false) + + // The session list's cache is warm, which is what a clear inside a running + // app leaves behind; the sweep reads whole rather than trusting it. + await newIndexer().start() + expect(sessionsMatching('indexed')).toEqual([SESSION_ID]) +}) + +it('refuses a reconcile before it is started and after it is closed', async () => { + newIndexer() + expect(() => indexer?.reconcile()).toThrow(/start\(\) first/) + + await indexer?.start() + await indexer?.reconcile() + indexer?.close() + expect(() => indexer?.reconcile()).toThrow(/closed/) +}) + +// I7: the sweep reads transcript bytes, so it stops at the same deadline every +// other pass does. It plans the whole machine and hands back what it had no +// time for; the passes that follow drain the plan without re-discovering. +it('stops the opening sweep at its deadline and drains the rest over the passes that follow', async () => { + for (let index = 0; index < 5; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`backlogged session ${index}`], session) + } + await newIndexer(readsPerPass(2)).start() + expect(indexer?.status().filesIndexed).toBe(2) + + await nextCycle() + expect(indexer?.status().filesIndexed).toBe(4) + + await nextCycle() + expect(sessionsMatching('backlogged')).toHaveLength(5) + expect(indexer?.status()).toMatchObject({ filesIndexed: 5, filesDue: 0 }) +}) + +// The sweep cadence, with nobody asking for it: a file outside the recency +// window that appears after the opening sweep is unreachable until the next +// periodic one, and the count of cycles is the whole rule. +it('sweeps on its cadence without anyone asking', async () => { + await writeClaudeTranscript(transcriptPath(), ['the newest conversation'], SESSION_ID) + await newIndexer({ recentPerAgent: 1, fullSweepEveryCycles: 2 }).start() + + const older = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(older, ['an older conversation'], OTHER_SESSION_ID) + const yesterday = new Date(Date.now() - 86_400_000) + await utimes(older, yesterday, yesterday) + + await nextCycle() + await nextCycle() + expect(sessionsMatching('older')).toEqual([]) + + await nextCycle() + expect(sessionsMatching('older')).toEqual([OTHER_SESSION_ID]) +}) + +// A cycle lists the newest N per agent, so every older row it holds is +// undiscovered and would be walked every twenty seconds. It proves the newest +// slice of them instead, capped: a transcript recent enough for the window is +// recent enough to be in the slice, and the rest are the next sweep's to reach. +// Round 12, F1. A directory that cannot be listed answers `unverifiable` for +// every row under it, on every pass, for as long as the permission stays wrong. +// With the walk capped at rows rather than at directories, five hundred such +// rows spent the whole budget on one readdir's worth of verdicts and a row for +// a file the user really deleted, sorted behind them, was never reached: six +// full sweeps and it was still held. +it.skipIf(!CAN_DENY_READ)('retires a deleted file behind a block of unreadable rows', async () => { + // A healthy project directory, so the root never looks emptied. + await writeClaudeTranscript(transcriptPath(), ['a live conversation'], SESSION_ID) + const locked = join(harness.roots.claudeProjectsDir ?? '', 'locked') + await mkdir(locked, { recursive: true }) + newIndexer() + + // What an unreadable tree leaves behind: rows the walk can never settle, + // planted ahead of the deleted one in the order the table returns them. + harness.write((db: SyncDatabase) => { + const insert = db.prepare( + `INSERT INTO files(path, byte_offset, mtime_ms, size_bytes, state) + VALUES (?, 0, ?, 10, 'current')` + ) + for (let index = 0; index < 520; index++) { + insert.run(join(locked, `locked-${index}.jsonl`), 1_700_000_000_000 + index) + } + return insert.run(join(harness.claudeProjectDir, 'deleted.jsonl'), 1_700_000_999_000) + }) + const deleted = join(harness.claudeProjectDir, 'deleted.jsonl') + const holdsDeleted = (): boolean => rowFor(deleted) !== undefined + + await chmod(locked, 0o000) + try { + await indexer?.start() + + expect(holdsDeleted()).toBe(false) + // And the block itself is neither retired nor forgotten: unreadable is not + // deleted, and the root is named as degraded rather than emptied. + expect(indexer?.status().filesIndexed).toBe(521) + expect(indexer?.status().phase).toBe('degraded') + } finally { + await chmod(locked, 0o700) + } +}) + +it('proves deletions for the newest rows it holds, and leaves the tail to a sweep', async () => { + const total = 530 + const oldest = transcriptPath('00000000-bbbb-4ccc-8ddd-eeeeeeeeeeee') + await writeClaudeTranscript( + oldest, + ['the oldest session'], + '00000000-bbbb-4ccc-8ddd-eeeeeeeeeeee' + ) + const longAgo = new Date(Date.now() - total * 60_000) + await utimes(oldest, longAgo, longAgo) + // Indexed on its own first, so it is the earliest row in the table as well as + // the oldest file. A slice that trusted the table's own order rather than the + // mtime would take it, and take it first. + await newIndexer().start() + + for (let index = 1; index < total; index++) { + const session = `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + const path = transcriptPath(session) + await writeClaudeTranscript(path, [`capped session ${index}`], session) + const at = new Date(Date.now() - (total - index) * 60_000) + await utimes(path, at, at) + } + await indexer?.reconcile({ full: true }) + expect(indexedSessionCount()).toBe(total) + + // Older than the cap reaches: 530 rows, twelve of them rediscovered by the + // cycle, leaves 518 undiscovered against a cap of 512. + await rm(oldest) + await nextCycle() + expect(indexedSessionCount()).toBe(total) + + await indexer?.reconcile({ full: true }) + expect(indexedSessionCount()).toBe(total - 1) +}) + +// F1: `fullSweepDue` stayed set across the sweep's await and was cleared on the +// way out, so a request raised while a sweep was running was erased by the +// sweep it arrived during. The pass takes the flag on entry now, and an +// unfinished sweep is what puts it back. +it('runs another sweep when one is asked for during a sweep', async () => { + for (let index = 0; index < 20; index++) { + const session = `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`recent session ${index}`], session) + } + const late = transcriptPath(OTHER_SESSION_ID) + + // Newest-one per root, so nothing but a second sweep can reach a file that + // appears after this sweep's discovery has already run. The clock is the one + // synchronous seam into a pass: it is read between files. + newIndexer({ recentPerAgent: 1 }) + let armed = false + // Once a row has landed the pass is provably inside its read loop, which is + // after it took the sweep flag and before it hands its verdicts back. + clock.onNow = () => { + if (armed || indexedSessionCount() === 0) { + return + } + armed = true + mkdirSync(dirname(late), { recursive: true }) + writeFileSync(late, `${claudeLines(['a late conversation'], OTHER_SESSION_ID, 0).join('\n')}\n`) + const backdated = new Date(Date.now() - 86_400_000) + utimesSync(late, backdated, backdated) + void indexer?.reconcile({ full: true }) + } + await indexer?.start() + await indexer?.settled() + + expect(sessionsMatching('late')).toEqual([OTHER_SESSION_ID]) +}) + +// The duty cycle, as a test: a pass reads for at most its deadline and hands +// the rest back, and the timer only re-arms once the pass has settled, so the +// share of the wall clock the index takes is bounded by construction. +it('hands the rest of a pass back when it runs out of wall time', async () => { + for (let index = 0; index < 20; index++) { + const session = `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`deadlined session ${index}`], session) + } + await newIndexer(readsPerPass(16)).start() + + expect(indexer?.status().filesIndexed).toBe(16) + + // And the pass after it picks up exactly the four it did not reach. + await nextCycle() + expect(indexer?.status()).toMatchObject({ filesIndexed: 20, filesDue: 0 }) +}) diff --git a/src/main/ai-vault-search/session-search-indexer.ts b/src/main/ai-vault-search/session-search-indexer.ts new file mode 100644 index 00000000000..26213242c49 --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer.ts @@ -0,0 +1,325 @@ +import { systemSessionSearchClock, type SessionSearchClock } from './session-search-clock' +import { + DEFAULT_SESSION_SEARCH_FULL_SWEEP_EVERY_CYCLES, + DEFAULT_SESSION_SEARCH_PASS_DEADLINE_FRACTION, + DEFAULT_SESSION_SEARCH_RECENT_PER_AGENT, + DEFAULT_SESSION_SEARCH_RECONCILE_INTERVAL_MS, + type SessionSearchIndexerOptions +} from './session-search-indexer-options' +import { SessionSearchDirectoryListings } from './session-search-directory-listings' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { runSessionSearchPass } from './session-search-pass' +import { sessionSearchHistoryCutoffMs } from './session-search-retention-policy' +import { SessionSearchStore, type SessionSearchStateCounts } from './session-search-store' +import type { SessionSearchDegradedRoot } from './session-search-degraded-roots' +import { SessionSearchWorkLoop } from './session-search-work-loop' + +/** + * Database paths a live indexer already owns. + * + * One process, one writer, one consumer registration per index. Two indexers on + * one path both register with the reader, so every transcript is read and + * written twice and the second write is fenced by the first at random. The + * recipe for every configuration change is close-then-construct, so the + * ordering that causes this is the one the recipe already rules out; this is + * what says so rather than letting it corrupt quietly. + */ +const liveIndexerPaths = new Set() + +export type SessionSearchIndexPhase = 'idle' | 'indexing' | 'current' | 'degraded' | 'closed' + +export type SessionSearchIndexStatus = { + phase: SessionSearchIndexPhase + /** Rows whose content matches the file at the stat the row records. */ + filesIndexed: number + /** Rows owed a whole read: a declined append, or a window that widened. */ + filesDue: number + /** Rows whose last read did not commit. */ + filesFailed: number + degradedRoots: SessionSearchDegradedRoot[] + lastReconcileAt: number | null + /** When a whole-machine sweep last finished; null until one has. */ + lastSweepCompletedAt: number | null +} + +/** + * Owns freshness for the index store: a whole-machine sweep, then a timer that + * keeps the newest N transcripts per agent reconciled and sweeps again every + * `fullSweepEveryCycles`. + * + * A library, not a service. It knows nothing about Electron, the app lifecycle, + * settings storage, IPC or the panel, and nothing here reads a setting or + * registers itself anywhere. Whoever constructs it decides all of that. + * + * **The store is the only memory.** Every question a pass asks between passes — + * what is owed a read, what has failed and how often, what the index holds and + * therefore what may have been deleted, what to report — is answered by a row + * in the `files` table. There is no queue, no watch set, no hold-out map and no + * counter with a reset rule. + * + * What is left here, and why none of it can be a row: + * - `previousRootsWithFiles`, the one bit per root the retirement walk's grace + * needs. Deliberately not durable: see the mountpoint trade in + * `session-search-deleted-sources.ts`. + * - `cyclesSinceSweep` and `sweepNext`, which are about the timer rather than + * about any file, and mean nothing to a second process. + * - `degradedRoots`, `lastReconcileAt` and `lastSweepCompletedAt`: what the last + * pass observed, held so `status()` can answer between passes. + * - `lastCounts`, the one cached query result, read only after `close()` so that + * describing what happened does not reopen a handle the owner has finished + * with. While the indexer is open every call re-queries. + * + * **Immutable after construction.** There is no `pause`, `resume`, `clear` or + * `setHistoryDays`. A configuration change is `close()` and a new instance; + * throwing the index away is + * `close(); removeSessionSearchDatabase(databasePath);` and a new instance. + * Widening retention is a new instance whose opening sweep admits the older + * files; narrowing is the purge that opens every full sweep. + * + * The guarantee it makes: while started, a transcript among the newest N per + * agent that grows, is replaced or is deleted is reflected in the index within + * one reconcile interval. Everything else is reached by the periodic sweep. + */ +export class SessionSearchIndexer { + private readonly ownershipPath: string + private readonly clock: SessionSearchClock + private readonly intervalMs: number + private readonly passDeadlineMs: number + private readonly recentPerAgent: number + private readonly fullSweepEveryCycles: number + private readonly onError: (error: unknown) => void + + private readonly loop: SessionSearchWorkLoop + private readonly store: SessionSearchStore + private readonly unregister: () => void + /** Null until a pass has recorded one; an empty set is a real observation. */ + private previousRootsWithFiles: ReadonlySet | null = null + private degradedRoots: SessionSearchDegradedRoot[] = [] + private lastReconcileAt: number | null = null + private lastSweepCompletedAt: number | null = null + private lastCounts: SessionSearchStateCounts | null = null + private cyclesSinceSweep = 0 + private sweepNext = false + private started = false + private closed = false + + constructor(private readonly options: SessionSearchIndexerOptions) { + this.ownershipPath = resolve(options.databasePath) + this.clock = options.clock ?? systemSessionSearchClock + this.intervalMs = options.reconcileIntervalMs ?? DEFAULT_SESSION_SEARCH_RECONCILE_INTERVAL_MS + this.passDeadlineMs = + options.passDeadlineMs ?? + Math.max(1, Math.floor(this.intervalMs / DEFAULT_SESSION_SEARCH_PASS_DEADLINE_FRACTION)) + this.recentPerAgent = options.recentPerAgent ?? DEFAULT_SESSION_SEARCH_RECENT_PER_AGENT + this.fullSweepEveryCycles = Math.max( + 1, + options.fullSweepEveryCycles ?? DEFAULT_SESSION_SEARCH_FULL_SWEEP_EVERY_CYCLES + ) + const onError = options.onError ?? ((error) => console.warn('[ai-vault-search]', error)) + this.onError = onError + this.loop = new SessionSearchWorkLoop({ + clock: this.clock, + intervalMs: this.intervalMs, + onFailure: onError + }) + if (liveIndexerPaths.has(this.ownershipPath)) { + throw new Error( + `SessionSearchIndexer: ${options.databasePath} already has a live indexer; close it first` + ) + } + // Store, registration and indexer share one lifetime, which is what makes + // the object immutable: there is no second open to get out of step with. + // Claimed only once the store is open, because a construction that throws + // has no `close()` to release the claim: registering first would leave the + // path owned by an object that does not exist, and every later attempt at + // it -- including the one that fixes whatever broke the open -- would be + // refused for the life of the process. + this.store = new SessionSearchStore(options.databasePath, onError) + liveIndexerPaths.add(this.ownershipPath) + this.store.setRetentionCutoffMs(this.cutoffMs()) + this.unregister = registerSessionSearchIndexConsumer(this.store) + } + + /** Runs a full sweep, then reconciles on the interval until closed. */ + start(): Promise { + if (this.closed || this.started) { + return this.loop.settled + } + this.started = true + this.sweepNext = true + return this.tick() + } + + /** + * Runs one pass now, off the timer. A full pass sweeps every root. + * + * Refused before `start()` and after `close()`: a pass against an indexer + * nobody started writes the index once and leaves it to go stale with no + * timer armed to notice the next change, and a pass against a closed one has + * no store to write to. Both are caller bugs, so both throw rather than + * resolving as though a pass had run. + */ + reconcile(options: { full?: boolean } = {}): Promise { + if (this.closed) { + throw new Error('SessionSearchIndexer.reconcile: the indexer is closed') + } + if (!this.started) { + throw new Error('SessionSearchIndexer.reconcile: start() first') + } + this.sweepNext ||= options.full === true + return this.tick() + } + + /** + * What the index holds, read from the rows rather than tallied. + * + * A second connection can compute every number here with one `GROUP BY`, + * which is the point: nothing is counted as it happens, so nothing can drift + * from what the database actually holds or need a rule about when to reset. + */ + status(): SessionSearchIndexStatus { + // A closed indexer reports what it last knew: opening a shut handle to + // answer a call whose whole job is to describe what happened is how a close + // came to report a database error to the owner who asked for it. + const settled = (this.closed ? this.lastCounts : this.readCounts()) ?? { + current: 0, + due: 0, + failed: 0 + } + return { + phase: this.phase(settled), + filesIndexed: settled.current, + filesDue: settled.due, + filesFailed: settled.failed, + degradedRoots: this.degradedRoots.map((root) => ({ ...root })), + lastReconcileAt: this.lastReconcileAt, + lastSweepCompletedAt: this.lastSweepCompletedAt + } + } + + /** Stops everything. Nothing queued before this call may run afterwards. */ + close(): void { + if (this.closed) { + return + } + // Read before the handle goes, so a status call afterwards reports what the + // index last held rather than opening a database its owner has finished with. + this.lastCounts = this.readCounts() ?? this.lastCounts + this.closed = true + // The loop, not just its timer: a task queued before this call would + // otherwise still run against a store this line is about to close. + this.loop.close() + this.unregister() + this.store.close() + liveIndexerPaths.delete(this.ownershipPath) + } + + /** Tests only: everything else drives this through the timer. */ + settled(): Promise { + return this.loop.settled + } + + private readCounts(): SessionSearchStateCounts | null { + try { + const counts = this.store.stateCounts() + this.lastCounts = counts + return counts + } catch (error) { + this.onError(error) + return this.lastCounts + } + } + + /** + * `current` is a claim, so it takes all three: no row owed a read, no row + * whose last read failed, and a whole sweep that finished. `idle` is the + * other end of it — an indexer nobody started has not promised to index + * anything, and calling that `current` would claim an index nobody built is + * up to date. + */ + private phase(counts: SessionSearchStateCounts): SessionSearchIndexPhase { + if (this.closed) { + return 'closed' + } + if (!this.started) { + return 'idle' + } + // A root the pass could not read, or a file it could not read: both are gaps + // the index knows about and cannot close on its own. + if (this.degradedRoots.length > 0 || counts.failed > 0) { + return 'degraded' + } + return counts.due === 0 && this.lastSweepCompletedAt !== null ? 'current' : 'indexing' + } + + private tick(): Promise { + return this.loop.queue( + (signal) => this.pass(signal), + () => void this.tick() + ) + } + + private async pass(signal: AbortSignal): Promise { + // The window moves with the clock, and the decide step reads it from the + // store. Setting it once at construction leaves a sweep purging rows that + // the very next candidate check happily re-indexes. + this.store.setRetentionCutoffMs(this.cutoffMs()) + // The one bound on a pass: wall time. What it does not reach is still owed, + // because a row says so and nothing had to be written down. + const startedAt = this.clock.now() + const full = this.sweepNext + // Taken on entry, not cleared on the way out: a `reconcile({ full: true })` + // raised while this pass is running sets it again, and clearing it at the + // end would erase that request along with this pass's own. + this.sweepNext = false + try { + const result = await runSessionSearchPass({ + store: this.store, + roots: this.options.roots, + full, + recentPerAgent: this.recentPerAgent, + previousRootsWithFiles: this.previousRootsWithFiles ?? undefined, + overdue: () => this.clock.now() - startedAt >= this.passDeadlineMs, + // One readdir per directory for the whole pass, shared by every step. + listings: new SessionSearchDirectoryListings(), + signal + }) + if (!result.completed) { + // A pass cut short learned nothing about root health, and publishing its + // empty findings would clear a live alarm. A sweep stays owed. + this.sweepNext ||= full + return + } + this.degradedRoots = result.degradedRoots + this.previousRootsWithFiles = result.rootsWithFiles + this.lastReconcileAt = this.clock.now() + // A backlog outside the recency window is only visible to a sweep, so a + // pass that ran out of time asks for one. It is self-limiting: the first + // pass that finishes its reads hands the interval back to cycles. + this.sweepNext ||= result.outOfTime + if (full) { + this.lastSweepCompletedAt = this.lastReconcileAt + this.cyclesSinceSweep = 0 + return + } + // A root that came back, a tree restored from a backup, an old transcript + // deleted: only a sweep sees any of it, and the count of cycles is the + // whole rule for when one is owed. + this.cyclesSinceSweep += 1 + if (this.cyclesSinceSweep >= this.fullSweepEveryCycles) { + this.sweepNext = true + } + } catch (error) { + // The flag is this method's to hold, so it is this method's to give back: + // a pass that threw part way learned nothing, and losing it here would + // leave nothing armed to try again. + this.sweepNext ||= full + throw error + } + } + + private cutoffMs(): number | null { + return sessionSearchHistoryCutoffMs(this.options.historyDays, this.clock.now()) + } +} +import { resolve } from 'node:path' diff --git a/src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts b/src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts new file mode 100644 index 00000000000..79cbe70eb7f --- /dev/null +++ b/src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts @@ -0,0 +1,378 @@ +import { chmod, mkdir, rename, rm } from 'node:fs/promises' +import { delimiter, dirname, join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import type { SessionSearchIndexerOptions } from './session-search-indexer-options' +import { removeSessionSearchDatabase } from './session-search-schema' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + writeMessageGraphTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +/* + * The lifecycle matrix: every operation a caller can perform, against every + * shape an unreachable root takes, against both ways discovery reports a root. + * + * The indexer is immutable, so "every operation" is a shorter list than it was: + * `pause`, `resume`, `clear`, `setHistoryDays` and `invalidate` are gone, and + * the two of them a caller still needs — a settings change and throwing the + * index away — are here as what replaced them, a new instance over the same + * path. In their place are the two passes the immutable design added: the + * periodic sweep, and a pass whose wall-clock deadline expires on its first file. + * + * What each cell asserts: + * A. No row is retired for a file that still exists. Throwing the index away + * is the one exception, and it is stated per operation rather than excused. + * B. The unreachable root is named in `degradedRoots`, by a real directory + * path — never the delimiter-joined label a merged discovery reports. + * C. The phase is never `current` while a root is degraded. + * D. Once the root is reachable again, a sweep indexes everything under it. + * + * Round 6 ran this as a throwaway harness on the previous design; it lives in + * the repository now. Two of its shapes changed with the stateless walk. The + * "present but empty mountpoint" shape is gone, because a readable root that + * lists nothing is no longer treated as unreachable — that is a root the user + * emptied, and `session-search-deleted-sources.ts` states the trade. In its + * place is a root whose transcripts sit behind an unreadable subdirectory, + * which is the partial-tree case the old shape never covered. + */ + +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const INTERVAL_MS = 20_000 +const SESSIONS = ['aaaaaaaa', 'bbbbbbbb', 'cccccccc'] + +type RootShape = { + name: string + /** Where the unreachable root's transcripts live, and where its files go. */ + detachedRoot: (harness: SessionSearchIndexerHarness) => string + detachedFile: (harness: SessionSearchIndexerHarness, session: string) => string + writeDetached: (path: string, session: string) => Promise + healthyFile: (harness: SessionSearchIndexerHarness, session: string) => string + writeHealthy: (path: string, session: string) => Promise +} + +const OPENCLAW_SESSION_DIR = join('agents', 'main', 'sessions') + +const ROOT_SHAPES: RootShape[] = [ + { + name: 'roots discovery reports one per directory', + detachedRoot: (harness) => harness.roots.claudeProjectsDir ?? '', + detachedFile: (harness, session) => join(harness.claudeProjectDir, `${session}.jsonl`), + writeDetached: (path, session) => + writeClaudeTranscript(path, [`detached ${session}`], fullSessionId(session)), + healthyFile: (harness, session) => join(harness.roots.piSessionsDir ?? '', `${session}.jsonl`), + writeHealthy: (path, session) => writeMessageGraphTranscript(path, [`healthy ${session}`]) + }, + { + name: 'roots a merged discovery joins into one label', + detachedRoot: (harness) => join(harness.roots.openclawStateDir ?? '', 'agents'), + detachedFile: (harness, session) => + join(harness.roots.openclawStateDir ?? '', OPENCLAW_SESSION_DIR, `${session}.jsonl`), + writeDetached: (path, session) => writeMessageGraphTranscript(path, [`detached ${session}`]), + healthyFile: (harness, session) => + join(harness.roots.openclawLegacyStateDir ?? '', OPENCLAW_SESSION_DIR, `${session}.jsonl`), + writeHealthy: (path, session) => writeMessageGraphTranscript(path, [`healthy ${session}`]) + } +] + +type UnreachableShape = { + name: string + needsDeniedRead: boolean + /** + * Whether an empty index can see this at all. Reading the root itself is the + * one probe a pass makes with no rows to go on: a root that answers ENOENT is + * what an uninstalled agent answers too, and a readable root with an + * unreadable subdirectory is swallowed by the file walker, which returns + * rather than reporting. Both are invisible until the index holds a row under + * the root, which is the evidence the retirement walk runs on. + */ + visibleWithNoRows: boolean + detach: (root: string, transcriptDir: string, parked: string) => Promise + attach: (root: string, transcriptDir: string, parked: string) => Promise +} + +const UNREACHABLE_SHAPES: UnreachableShape[] = [ + { + name: 'the root itself is not there', + needsDeniedRead: false, + visibleWithNoRows: false, + detach: (root, _transcriptDir, parked) => rename(root, parked), + attach: (root, _transcriptDir, parked) => rename(parked, root) + }, + { + name: 'the root refuses to list', + needsDeniedRead: true, + visibleWithNoRows: true, + detach: (root) => chmod(root, 0o000), + attach: (root) => chmod(root, 0o755) + }, + { + name: 'the transcripts sit behind a directory that refuses to list', + needsDeniedRead: true, + visibleWithNoRows: false, + detach: (_root, transcriptDir) => chmod(transcriptDir, 0o000), + attach: (_root, transcriptDir) => chmod(transcriptDir, 0o755) + } +] + +type Operation = { + name: string + /** True when the operation throws the index away, so no row survives it. */ + clearsIndex?: boolean + /** Healthy-root sessions the operation deletes from disk. */ + deletes?: readonly string[] + /** Construction options for every indexer this cell opens. */ + options?: Partial + run: (context: MatrixContext) => Promise +} + +const OPERATIONS: Operation[] = [ + { name: 'one cycle', run: (context) => context.cycle() }, + { + name: 'two cycles', + run: async (context) => { + await context.cycle() + await context.cycle() + } + }, + { + name: 'close and restart', + run: (context) => context.reopen() + }, + { + name: 'two full reconciles', + run: async (context) => { + await context.indexer().reconcile({ full: true }) + await context.indexer().reconcile({ full: true }) + } + }, + { + name: 'one healthy transcript deleted', + deletes: SESSIONS.slice(0, 1), + run: (context) => context.cycle() + }, + { + name: 'every healthy transcript deleted', + deletes: SESSIONS, + run: async (context) => { + // Twice: a root that goes from holding transcripts to holding none in one + // pass is unverifiable for that pass, so the second is the proving one. + await context.indexer().reconcile({ full: true }) + await context.indexer().reconcile({ full: true }) + } + }, + { + // The cadence that replaced every re-arm-on-recovery rule: no caller asks + // for this sweep, so the cell drives it off the timer alone. + name: 'the periodic sweep comes round', + options: { fullSweepEveryCycles: 2 }, + run: async (context) => { + await context.cycle() + await context.cycle() + await context.cycle() + } + }, + { + // Every pass is out of wall time from its first file, so each one hands + // almost all of its work back. A pass that read almost nothing must still + // not conclude anything about what it did not reach. + name: 'every pass out of time at its first file', + options: { passDeadlineMs: 0 }, + run: async (context) => { + await context.cycle() + await context.cycle() + } + }, + { + // What replaced `setHistoryDays`: a new instance over the same database. + // Every transcript here was written just now, so a 30-day window holds all + // of them and no row may be purged. + name: 'reconstructed for a narrower history window', + run: (context) => context.reopen({ historyDays: 30 }) + }, + { + // What replaced `clear()`, exactly as the PR body documents it. + name: 'the index thrown away and rebuilt', + clearsIndex: true, + run: (context) => context.reopen({ removeDatabase: true }) + } +] + +type MatrixContext = { + indexer: () => SessionSearchIndexer + /** Closes and constructs again over the same path: the immutable design's one edit. */ + reopen: (args?: { historyDays?: number | null; removeDatabase?: boolean }) => Promise + cycle: () => Promise + detachedRoot: string + detachedPaths: string[] +} + +function fullSessionId(prefix: string): string { + return `${prefix}-bbbb-4ccc-8ddd-eeeeeeeeeeee` +} + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-lifecycle') + indexer = null +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function open(overrides: Partial = {}): SessionSearchIndexer { + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS, + ...overrides + }) + return indexer +} + +/** + * Runs cycles until the index stops growing. Every operation but the + * out-of-time one settles on the first call; that one reads a transcript a pass. + */ +async function driveUntilIndexed(maxCycles: number): Promise { + let held = indexedSessions().length + for (let cycle = 0; cycle < maxCycles; cycle++) { + clock.advance(INTERVAL_MS) + await indexer?.settled() + const now = indexedSessions().length + if (now === held) { + return + } + held = now + } +} + +/** Session ids the index answers for, whichever agent wrote them. */ +function indexedSessions(): string[] { + return harness + .read( + (db: SyncDatabase) => + db.prepare('SELECT session_id AS id FROM sessions').all() as { id: string }[] + ) + .map((row) => row.id) + .sort() +} + +for (const roots of ROOT_SHAPES) { + for (const unreachable of UNREACHABLE_SHAPES) { + describe.skipIf(unreachable.needsDeniedRead && !CAN_DENY_READ)( + `${roots.name}, ${unreachable.name}`, + () => { + for (const operation of OPERATIONS) { + it(operation.name, async () => { + const detachedRoot = roots.detachedRoot(harness) + const detachedPaths = SESSIONS.map((session) => roots.detachedFile(harness, session)) + const healthyPaths = SESSIONS.map((session) => roots.healthyFile(harness, session)) + for (const [index, session] of SESSIONS.entries()) { + await roots.writeDetached(detachedPaths[index] ?? '', session) + await roots.writeHealthy(healthyPaths[index] ?? '', session) + } + const transcriptDir = dirname(detachedPaths[0] ?? '') + const parked = join(harness.root, 'parked-root') + + await open(operation.options).start() + // A deadline that expires on the first file reads one transcript a + // pass, so the setup drives passes until the index has caught up. + await driveUntilIndexed(SESSIONS.length * 2) + const detachedIds = detachedPaths.map((_path, index) => + roots === ROOT_SHAPES[0] + ? fullSessionId(SESSIONS[index] ?? '') + : (SESSIONS[index] ?? '') + ) + const healthyIds = SESSIONS.map((session) => session) + expect(indexedSessions()).toEqual([...detachedIds, ...healthyIds].sort()) + // One cycle so the watch set holds the recency window, which is the + // state a running indexer is in when a volume goes away. + clock.advance(INTERVAL_MS) + await indexer?.settled() + + await unreachable.detach(detachedRoot, transcriptDir, parked) + try { + const kept = SESSIONS.filter((session) => !operation.deletes?.includes(session)) + for (const session of operation.deletes ?? []) { + await rm(healthyPaths[SESSIONS.indexOf(session)] ?? '') + } + await operation.run({ + indexer: () => indexer as SessionSearchIndexer, + reopen: async (args = {}) => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + if (args.removeDatabase === true) { + removeSessionSearchDatabase(harness.databasePath) + } + const overrides = { ...operation.options } + if ('historyDays' in args) { + overrides.historyDays = args.historyDays + } + await open(overrides).start() + await driveUntilIndexed(SESSIONS.length * 2) + }, + cycle: async () => { + clock.advance(INTERVAL_MS) + await indexer?.settled() + }, + detachedRoot, + detachedPaths + }) + + // A: nothing that still exists lost its rows. + const survivingDetached = operation.clearsIndex ? [] : detachedIds + expect(indexedSessions()).toEqual([...survivingDetached, ...kept].sort()) + + const status = indexer?.status() + const degraded = status?.degradedRoots.map((root) => root.root) ?? [] + // With no rows under it, the only thing a pass can go on is + // whether the root itself refuses to list. + if (operation.clearsIndex && !unreachable.visibleWithNoRows) { + expect(degraded).not.toContain(detachedRoot) + } else { + // B: named, by a real directory rather than a joined label. + expect(degraded).toContain(detachedRoot) + expect(degraded.every((root) => !root.includes(delimiter))).toBe(true) + // C: not current while a root is degraded. + expect(status?.phase).not.toBe('current') + } + } finally { + await unreachable.attach(detachedRoot, transcriptDir, parked) + } + + // D: reachable again, a sweep reads the whole tree back. + await mkdir(dirname(healthyPaths[0] ?? ''), { recursive: true }) + await indexer?.reconcile({ full: true }) + await driveUntilIndexed(SESSIONS.length * 2) + expect(indexedSessions()).toEqual( + [ + ...detachedIds, + ...SESSIONS.filter((session) => !operation.deletes?.includes(session)) + ].sort() + ) + }) + } + } + ) + } +} diff --git a/src/main/ai-vault-search/session-search-live-transcript.test.ts b/src/main/ai-vault-search/session-search-live-transcript.test.ts index 7f37ca662cb..1ea58ca5283 100644 --- a/src/main/ai-vault-search/session-search-live-transcript.test.ts +++ b/src/main/ai-vault-search/session-search-live-transcript.test.ts @@ -193,16 +193,16 @@ it('indexes a file the session list already read past, once a whole read is aske // The append continued from a byte offset the index never saw, so it declined. expect(sessionsMatching('zygomorphic')).toEqual([]) - const behind = store.takeStale() - expect(behind.map((candidate) => candidate.file.path)).toEqual([path]) - for (const candidate of behind) { - requestWholeTranscriptRead(candidate.file.path) - } + // The index holds no row for this file at all, and that is the record: a + // path the file table does not name is read from the start by the next pass, + // which is what asks the reader to drop the session list's resume point. + expect(store.files()).toEqual([]) + requestWholeTranscriptRead(path) const reread = await parseTranscript(path) expect(reread.stats).toMatchObject({ incremental: 0, fullParses: 1 }) expect(errors).toEqual([]) expect(sessionsMatching('zygomorphic')).toEqual([SESSION_ID]) expect(sessionsMatching('opening')).toEqual([SESSION_ID]) - expect(store.takeStale()).toEqual([]) + expect(store.files().map((row) => row.state)).toEqual(['current']) }) diff --git a/src/main/ai-vault-search/session-search-merged-roots.test.ts b/src/main/ai-vault-search/session-search-merged-roots.test.ts new file mode 100644 index 00000000000..8d41b047cae --- /dev/null +++ b/src/main/ai-vault-search/session-search-merged-roots.test.ts @@ -0,0 +1,135 @@ +import { chmod, rm } from 'node:fs/promises' +import { delimiter, join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeMessageGraphTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +// OpenClaw is the one agent whose roots are alternates for a single install, so +// discovery reports them as ONE discovery whose rootDir is every path joined by +// the platform's path delimiter. That string is not a directory: readdir on it +// answers ENOENT, containment never matches a real file, and a scan issue +// recorded against a real root never compares equal to it. Everything that +// judges a root works on the constituent directories, taken from the same +// source table discovery reads, never by splitting the label -- a directory may +// legally contain the delimiter. + +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const INTERVAL_MS = 20_000 + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-merged-roots') +}) + +afterEach(async () => { + indexer.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +/** OpenClaw reads `/agents/**` and keeps only paths through `sessions`. */ +function openclawTranscript(stateDir: string, name: string): string { + return join(stateDir, 'agents', 'main', 'sessions', `${name}.jsonl`) +} + +function sessionsMatching(term: string): string[] { + return harness.read((db: SyncDatabase) => + ( + db + .prepare( + `SELECT DISTINCT s.session_id AS id FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY s.session_id` + ) + .all(term) as { id: string }[] + ).map((row) => row.id) + ) +} + +it.skipIf(!CAN_DENY_READ)('fences one merged root without taking its partner down', async () => { + const current = harness.roots.openclawStateDir ?? '' + const legacy = harness.roots.openclawLegacyStateDir ?? '' + const mounted = openclawTranscript(current, 'mounted-session') + const local = openclawTranscript(legacy, 'local-session') + await writeMessageGraphTranscript(mounted, ['a conversation on the mounted volume']) + await writeMessageGraphTranscript(local, ['a conversation on local disk']) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS + }) + await indexer.start() + expect(sessionsMatching('conversation').sort()).toEqual(['local-session', 'mounted-session']) + + // One of the two roots goes away; the other is untouched. + await chmod(join(current, 'agents'), 0o000) + try { + await indexer.reconcile({ full: true }) + + const status = indexer.status() + const degraded = status.degradedRoots.map((root) => root.root) + // A real directory, not the joined string discovery reports. + expect(degraded).toContain(join(current, 'agents')) + expect(degraded.every((root) => !root.includes(delimiter))).toBe(true) + // Unprovable, so the unreadable root keeps its rows. + expect(sessionsMatching('mounted')).toEqual(['mounted-session']) + } finally { + await chmod(join(current, 'agents'), 0o755) + } +}) + +it('retires from one merged root while its partner is healthy', async () => { + const current = harness.roots.openclawStateDir ?? '' + const legacy = harness.roots.openclawLegacyStateDir ?? '' + const going = openclawTranscript(current, 'going-session') + await writeMessageGraphTranscript(going, ['a conversation about to be deleted']) + // A sibling in the same root, so deleting one leaves the root listing files + // and therefore healthy: this is a deletion, not an unmount. + await writeMessageGraphTranscript(openclawTranscript(current, 'sibling-session'), [ + 'a conversation beside it' + ]) + await writeMessageGraphTranscript(openclawTranscript(legacy, 'staying-session'), [ + 'a conversation that stays' + ]) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS + }) + await indexer.start() + expect(sessionsMatching('conversation').sort()).toEqual([ + 'going-session', + 'sibling-session', + 'staying-session' + ]) + + // A genuine deletion inside a healthy root still retires normally. + await rm(going) + await indexer.reconcile({ full: true }) + + expect(sessionsMatching('deleted')).toEqual([]) + expect(indexer.status().degradedRoots).toEqual([]) + expect(sessionsMatching('conversation').sort()).toEqual(['sibling-session', 'staying-session']) +}) diff --git a/src/main/ai-vault-search/session-search-native-chat-indexing.test.ts b/src/main/ai-vault-search/session-search-native-chat-indexing.test.ts new file mode 100644 index 00000000000..deab4e8b785 --- /dev/null +++ b/src/main/ai-vault-search/session-search-native-chat-indexing.test.ts @@ -0,0 +1,113 @@ +import { appendFile, mkdir, writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +// Reviewer F4, and the plan's fourth open decision: a conversation held in +// Orca's own chat is the same file in the same place as one held in the +// terminal, so it must be searchable through the same path with no panel +// mounted, no scanner service running, and nobody calling refresh. Everything +// below is the library and the filesystem. + +const INTERVAL_MS = 20_000 +const SESSION_ID = 'cccccccc-dddd-4eee-8fff-000000000000' +const CWD = '/repo/orca' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-native-chat') +}) + +afterEach(async () => { + indexer.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +/** The rows Orca's native chat writes: uuid, block content, cwd on the first turn. */ +function nativeChatTurn(uuid: string, role: 'user' | 'assistant', text: string): string { + const timestamp = new Date(1_740_000_000_000 + Number(uuid.slice(-2)) * 60_000).toISOString() + return JSON.stringify({ + type: role, + uuid, + sessionId: SESSION_ID, + timestamp, + cwd: CWD, + gitBranch: 'main', + message: { + role, + ...(role === 'assistant' ? { model: 'claude-fable-5' } : {}), + content: [{ type: 'text', text }] + } + }) +} + +function messageTexts(term: string): { role: string; session: string }[] { + return harness.read( + (db: SyncDatabase) => + db + .prepare( + `SELECT m.role AS role, s.session_id AS session FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY m.id` + ) + .all(term) as { role: string; session: string }[] + ) +} + +it('indexes a native-chat conversation and its later turns with no panel and no service', async () => { + const path = join(harness.claudeProjectDir, `${SESSION_ID}.jsonl`) + await mkdir(harness.claudeProjectDir, { recursive: true }) + await writeFile( + path, + `${[ + nativeChatTurn('turn-01', 'user', 'why does the relay drop the lease at 105 seconds'), + nativeChatTurn('turn-02', 'assistant', 'that is the client silence watchdog, not a cliff') + ].join('\n')}\n` + ) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS + }) + await indexer.start() + + expect(messageTexts('watchdog')).toEqual([{ role: 'assistant', session: SESSION_ID }]) + expect(harness.read((db: SyncDatabase) => db.prepare('SELECT cwd FROM sessions').get())).toEqual({ + cwd: CWD + }) + + // The conversation continues in the panel; nothing tells the index about it. + await appendFile( + path, + `${[ + nativeChatTurn('turn-03', 'user', 'and the fleetwide 4408 bursts'), + nativeChatTurn('turn-04', 'assistant', 'those are desktop lease rotations, cohort waves') + ].join('\n')}\n` + ) + clock.advance(INTERVAL_MS) + await indexer.settled() + + expect(messageTexts('cohort')).toEqual([{ role: 'assistant', session: SESSION_ID }]) + expect(messageTexts('4408')).toEqual([{ role: 'user', session: SESSION_ID }]) + expect(indexer.status().phase).toBe('current') +}) diff --git a/src/main/ai-vault-search/session-search-opencode-decline.test.ts b/src/main/ai-vault-search/session-search-opencode-decline.test.ts new file mode 100644 index 00000000000..1ed3a201727 --- /dev/null +++ b/src/main/ai-vault-search/session-search-opencode-decline.test.ts @@ -0,0 +1,177 @@ +import { mkdirSync } from 'node:fs' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' + +// Only the thread hop is replaced: both implementations below are the repo's +// own in-process readers, which the worker entry calls on the other side. +export const openCodeParseCalls: string[] = [] +vi.mock('../ai-vault/session-scanner-opencode-sqlite-worker-spawn', async () => { + const list = await import('../ai-vault/session-scanner-opencode-sqlite-list') + const parse = await import('../ai-vault/session-scanner-opencode-sqlite') + const own = await import('./session-search-opencode-decline.test') + return { + resolveOpenCodeSqliteWorkerEntryPath: () => null, + listOpenCodeSqliteSessionsViaWorker: ( + args: Parameters[0] + ) => list.listOpenCodeSqliteSessions(args), + parseOpenCodeSqliteSessionViaWorker: ( + args: Parameters[0] + ) => { + own.openCodeParseCalls.push(args.sessionId) + return parse.parseOpenCodeSqliteSession(args) + } + } +}) +import Database from '../sqlite/sync-database' +import { getSessionParseCacheEntry } from '../ai-vault/session-parse-cache-store' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { buildOpenCodeSqliteCandidatePath } from '../ai-vault/session-scanner-opencode-sqlite-paths' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +/* + * Round 12, F3. An OpenCode SQLite session decodes where the message channel + * cannot reach it, so no read of one will ever commit a row. The consumer + * declined it and wrote nothing, which left the file table silent about a + * source discovery returns on every pass: the decide step saw a path the index + * held nothing for, asked for a read, and asking for one over a warm cache + * drops the session list's own resume point. Every OpenCode session was fully + * decoded on every pass and the sidebar's fold was thrown away with it, which + * is the cache STA-1278 and STA-1417 added. + */ + +const SESSION = 'ses_r12' +const CLAUDE_SESSION = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null = null + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-opencode-decline') + indexer = null + openCodeParseCalls.length = 0 +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function writeOpenCodeDb(path: string, sessionId: string): void { + const db = new Database(path) + db.exec(` + CREATE TABLE session ( + id TEXT PRIMARY KEY, project_id TEXT NOT NULL, parent_id TEXT, slug TEXT NOT NULL, + directory TEXT NOT NULL, title TEXT NOT NULL, version TEXT NOT NULL, share_url TEXT, + summary_additions INTEGER, summary_deletions INTEGER, summary_files INTEGER, + summary_diffs TEXT, revert TEXT, permission TEXT, + time_created INTEGER NOT NULL, time_updated INTEGER NOT NULL, time_compacting INTEGER, + time_archived INTEGER, workspace_id TEXT, path TEXT, agent TEXT, model TEXT, + cost REAL DEFAULT 0 NOT NULL, tokens_input INTEGER DEFAULT 0 NOT NULL, + tokens_output INTEGER DEFAULT 0 NOT NULL, tokens_reasoning INTEGER DEFAULT 0 NOT NULL, + tokens_cache_read INTEGER DEFAULT 0 NOT NULL, tokens_cache_write INTEGER DEFAULT 0 NOT NULL, + metadata TEXT + ); + CREATE TABLE message ( + id TEXT PRIMARY KEY, session_id TEXT NOT NULL, time_created INTEGER NOT NULL, + time_updated INTEGER NOT NULL, data TEXT NOT NULL + ); + CREATE TABLE project ( + id TEXT PRIMARY KEY, worktree TEXT NOT NULL, vcs TEXT, name TEXT, icon_url TEXT, + icon_color TEXT, time_created INTEGER NOT NULL, time_updated INTEGER NOT NULL, + time_initialized INTEGER, sandboxes TEXT NOT NULL, commands TEXT, icon_url_override TEXT + ); + CREATE TABLE part ( + id TEXT PRIMARY KEY, message_id TEXT NOT NULL, session_id TEXT NOT NULL, + time_created INTEGER NOT NULL, time_updated INTEGER NOT NULL, data TEXT NOT NULL + ); + `) + db.prepare( + `INSERT INTO session (id, project_id, parent_id, slug, directory, title, version, + time_created, time_updated, agent, model, cost, tokens_input, tokens_output, + tokens_reasoning, tokens_cache_read, tokens_cache_write) + VALUES (?, 'proj-1', NULL, 'slug-1', '/tmp/opencode', 'OpenCode title', '1.0.0', + ?, ?, 'build', '{"id":"glm"}', 0, 1, 1, 0, 0, 0)` + ).run(sessionId, 1_740_000_000_000, 1_740_000_100_000) + db.prepare( + `INSERT INTO message (id, session_id, time_created, time_updated, data) VALUES (?, ?, ?, ?, ?)` + ).run( + 'msg-1', + sessionId, + 1_740_000_000_000, + 1_740_000_000_000, + JSON.stringify({ role: 'user', time: { created: 1_740_000_000_000 } }) + ) + db.prepare( + `INSERT INTO part (id, message_id, session_id, time_created, time_updated, data) VALUES (?, ?, ?, ?, ?, ?)` + ).run( + 'part-1', + 'msg-1', + sessionId, + 1_740_000_000_000, + 1_740_000_000_000, + JSON.stringify({ type: 'text', text: 'hello opencode' }) + ) + db.prepare( + `INSERT INTO project (id, worktree, name, time_created, time_updated, sandboxes) + VALUES ('proj-1', '/tmp/opencode', 'proj', ?, ?, '[]')` + ).run(1_740_000_000_000, 1_740_000_000_000) + db.close() +} + +it('reads an OpenCode session once, not on every pass', async () => { + const dbPath = join(harness.root, 'opencode-db', 'opencode.db') + mkdirSync(join(harness.root, 'opencode-db'), { recursive: true }) + writeOpenCodeDb(dbPath, SESSION) + const claudePath = join(harness.claudeProjectDir, 'control.jsonl') + await writeClaudeTranscript(claudePath, ['control turn'], CLAUDE_SESSION) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: { ...harness.roots, opencodeDbPaths: [dbPath] }, + historyDays: null, + clock, + reconcileIntervalMs: 20_000, + onError: () => undefined + }) + await indexer.start() + + const syntheticPath = buildOpenCodeSqliteCandidatePath(dbPath, SESSION) + const openCodeAfterFirst = getSessionParseCacheEntry(syntheticPath) + const claudeAfterFirst = getSessionParseCacheEntry(claudePath) + + await indexer.reconcile() + await indexer.reconcile() + + // One decode across three passes, and the session list's cached fold for it + // is the same object it was after the first: nothing invalidated it. + expect(openCodeParseCalls).toHaveLength(1) + expect(getSessionParseCacheEntry(syntheticPath)).toBe(openCodeAfterFirst) + // The control, which the index really does hold, is untouched either way. + expect(getSessionParseCacheEntry(claudePath)).toBe(claudeAfterFirst) + + // What makes it skippable: a row saying the index has seen this source and + // holds no session for it, which is the shape a read-through-with-no-session + // already leaves. + const rows = harness.read((db) => + db.prepare('SELECT path, state, session_row_id FROM files ORDER BY path').all() + ) as { path: string; state: string; session_row_id: number | null }[] + expect(rows).toHaveLength(2) + expect(rows.find((row) => row.path === syntheticPath)).toMatchObject({ + state: 'current', + session_row_id: null + }) + expect(indexer.status()).toMatchObject({ filesDue: 0, filesFailed: 0, phase: 'current' }) +}) diff --git a/src/main/ai-vault-search/session-search-pass.ts b/src/main/ai-vault-search/session-search-pass.ts new file mode 100644 index 00000000000..d4f36015226 --- /dev/null +++ b/src/main/ai-vault-search/session-search-pass.ts @@ -0,0 +1,216 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import { ensureSessionParseCacheLoaded } from '../ai-vault/session-parse-cache-persistence' +import { + cursorChatMetaRefusals, + withCursorChatMetaScan +} from '../ai-vault/session-scanner-cursor-chat-meta' +import { recordSessionScanIssue } from '../ai-vault/session-scan-issues' +import { + mergeDegradedRoots, + scanIssueDegradedRoots, + unreadableRoots, + type SessionSearchDegradedRoot +} from './session-search-degraded-roots' +import { retireDeletedSessionSearchSources } from './session-search-deleted-sources' +import type { SessionSearchDirectoryReader } from './session-search-directory-listings' +import { runSessionSearchIndexPass } from './session-search-index-pass' +import { + discoverSessionSearchCandidates, + isUnderScanRoot, + sessionSearchEmptiedRoots, + sessionSearchRootListings, + type SessionSearchScanRoots +} from './session-search-scan-roots' +import type { SessionSearchFileRow, SessionSearchStore } from './session-search-store' +import { sessionSearchEnumeratedContainers } from './session-search-synthetic-sources' + +/** + * Rows a cycle proves present or gone, newest first. + * + * Why bounded and why newest first: a cycle lists the newest N per agent, so + * every older row it holds is undiscovered and would otherwise be walked every + * twenty seconds. Newest first is what makes the guarantee hold — a transcript + * recent enough for the window to cover is recent enough to be in this slice, + * so its deletion is proven on the very next cycle whenever it happened. + */ +const RETIREMENT_ROWS_PER_CYCLE = 512 + +/** + * Directories either pass may read proving deletions. + * + * The bound on the walk is readdirs, not rows: rows sharing a directory are one + * read and then map lookups, and a directory that answers an error answers it + * once for every row under it. Counting rows instead let one unreadable + * directory hold the whole walk for as long as it stayed unreadable. + */ +const RETIREMENT_DIRECTORIES_PER_PASS = 512 + +export type SessionSearchPassArgs = { + store: SessionSearchStore + roots: SessionSearchScanRoots + /** A sweep lists every root; a cycle lists the newest N per agent. */ + full: boolean + recentPerAgent: number + /** Real roots that listed transcripts on the previous pass; undefined before the first. */ + previousRootsWithFiles?: ReadonlySet + /** True once the pass is out of wall time; reads stop, everything else finishes. */ + overdue?: () => boolean + /** One readdir per directory for the whole pass, shared by every step. */ + listings: SessionSearchDirectoryReader + signal?: AbortSignal +} + +export type SessionSearchPassResult = { + /** Real roots this pass listed transcripts under, for the next pass to compare against. */ + rootsWithFiles: Set + degradedRoots: SessionSearchDegradedRoot[] + /** False when the pass was cut short; its conclusions are not to be recorded. */ + completed: boolean + /** + * True when the deadline stopped the reads with candidates still owed. + * + * The caller's one use for it: a cycle lists the newest N per agent, so a + * backlog outside that window is only *visible* to a sweep. Without this a + * first run would index the recency window in its opening pass and then crawl, + * making progress only on the periodic sweep every five minutes. + */ + outOfTime: boolean +} + +/** + * One pass. Four steps, the same four whether it sweeps or cycles. + * + * 1. **Discover.** The only filesystem walk: every root on a sweep, the newest + * N per agent on a cycle. Everything below is decided from what it returns. + * 2. **Decide and read.** Per candidate, its stat against its row. Reads stop + * at the deadline and nothing is recorded about what was left, because being + * owed is a fact about the row and not an entry in a queue. + * 3. **Retire.** Candidates are the rows this pass's discovery did not return, + * inside the scope that discovery covered. The stateless walk proves each + * one gone, present or unverifiable; only `gone` deletes. + * 4. **Report.** Root health for this pass. The counts are a query, made by the + * caller against the same rows, so nothing here is tallied. + * + * The pass keeps nothing. Everything it learns is either on a row or in the + * result the caller compares against the next pass. + */ +export async function runSessionSearchPass( + args: SessionSearchPassArgs +): Promise { + const { store, signal } = args + if (args.full) { + // Every sweep opens with the purge, so a window narrower than the last + // instance held is applied by the first sweep of this one. + await store.purgeOlderThan(store.retentionCutoff, signal) + } + await ensureSessionParseCacheLoaded() + return withCursorChatMetaScan(async () => { + const swept = await discoverSessionSearchCandidates(args.roots, { + limitPerAgent: args.full ? Number.POSITIVE_INFINITY : args.recentPerAgent, + signal + }) + const issues: AiVaultScanIssue[] = [...swept.issues] + + let completed = true + let outOfTime = false + const rows = new Map(store.files().map((row) => [row.path, row])) + try { + const read = await runSessionSearchIndexPass(store, swept.candidates, { + signal, + rows, + overdue: args.overdue + }) + outOfTime = read.outOfTime + } catch (error) { + if (!signal?.aborted) { + throw error + } + completed = false + } + + const listings = sessionSearchRootListings(args.roots, swept.discoveries) + const roots = listings.map((listing) => listing.root) + const rootsWithFiles = new Set( + listings.filter((listing) => listing.files > 0).map((listing) => listing.root) + ) + // Undefined, not empty, before any pass has recorded one: an empty set is a + // real observation and this is the absence of one. + const previousRootsWithFiles = args.previousRootsWithFiles + // A pass cut short saw part of the machine, so its silence about a path is + // not evidence; it retires nothing and publishes no verdicts. + const retirement = completed + ? await retireDeletedSessionSearchSources({ + store, + paths: retirementCandidates(rows, swept, roots, args.full), + roots, + // Only a sweep enumerates without a per-agent limit, so only a sweep + // may prove a synthetic row's container holds it no longer. + enumeratedContainers: args.full + ? sessionSearchEnumeratedContainers(swept.candidates, issues) + : undefined, + emptiedRoots: previousRootsWithFiles + ? sessionSearchEmptiedRoots(previousRootsWithFiles, rootsWithFiles) + : new Set(), + listings: args.listings, + directoryLimit: RETIREMENT_DIRECTORIES_PER_PASS, + signal + }) + : { retired: [], unverifiable: [], unchecked: [], degradedRoots: [] } + + for (const refusal of cursorChatMetaRefusals()) { + // One issue per refused chats root, not one per Cursor transcript. + recordSessionScanIssue(issues, { + agent: 'cursor', + path: refusal.chatsRoot, + message: refusal.message + }) + } + // Roots that listed no transcripts and cannot be listed either: the walker + // swallows a readdir failure, so this is the only place it surfaces. + const unlistable = completed + ? await unreadableRoots( + roots.filter((root) => !rootsWithFiles.has(root)), + args.listings, + signal + ) + : [] + + return { + rootsWithFiles, + degradedRoots: mergeDegradedRoots( + scanIssueDegradedRoots(roots, issues), + retirement.degradedRoots, + unlistable + ), + completed, + outOfTime + } + }) +} + +/** + * Rows this pass's discovery did not return, inside the scope it covered. + * + * A sweep covers everything, so every undiscovered row is a candidate. A cycle + * covers the newest N per agent, so it may only judge rows under a root it + * actually listed, and it takes the newest of those: an older row is not + * evidence of anything a cycle looked for, and the next sweep is what reaches + * it. This is the whole of what used to be a watch set carried between passes. + */ +function retirementCandidates( + rows: ReadonlyMap, + swept: { candidates: readonly { file: { path: string } }[] }, + roots: readonly string[], + full: boolean +): string[] { + const discovered = new Set(swept.candidates.map((candidate) => candidate.file.path)) + const undiscovered = [...rows.values()].filter((row) => !discovered.has(row.path)) + if (full) { + return undiscovered.map((row) => row.path) + } + return undiscovered + .filter((row) => roots.some((root) => isUnderScanRoot(row.path, root))) + .sort((left, right) => right.mtimeMs - left.mtimeMs) + .slice(0, RETIREMENT_ROWS_PER_CYCLE) + .map((row) => row.path) +} diff --git a/src/main/ai-vault-search/session-search-read-decision.test.ts b/src/main/ai-vault-search/session-search-read-decision.test.ts new file mode 100644 index 00000000000..64eb8866d02 --- /dev/null +++ b/src/main/ai-vault-search/session-search-read-decision.test.ts @@ -0,0 +1,117 @@ +import { expect, it } from 'vitest' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type { SessionSearchIndexedFile } from './session-search-file-cursor' +import { + SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT, + sessionSearchReadDecision +} from './session-search-read-decision' +import type { SessionSearchFileRow } from './session-search-store' + +const PATH = '/transcripts/one.jsonl' +const MTIME = 1_740_000_000_000 + +function candidate(overrides: Partial = {}): SessionFileCandidate { + return { + agent: 'claude', + codexHome: null, + file: { + path: PATH, + mtimeMs: MTIME, + modifiedAt: new Date(MTIME).toISOString(), + sizeBytes: 100, + ...overrides + } + } +} + +function row(overrides: Partial = {}): SessionSearchFileRow { + return { + path: PATH, + identity: null, + mtimeMs: MTIME, + sizeBytes: 100, + state: 'current', + failCount: 0, + failedMtimeMs: null, + ...overrides + } +} + +const cursor: SessionSearchIndexedFile = { byteOffset: 100, mtimeMs: MTIME, sizeBytes: 100 } + +function decide(args: { + file?: Partial + row?: SessionSearchFileRow | undefined + cursor?: SessionSearchIndexedFile | null + cutoffMs?: number | null +}) { + return sessionSearchReadDecision({ + candidate: candidate(args.file), + row: 'row' in args ? args.row : row(), + cursor: 'cursor' in args ? (args.cursor ?? null) : cursor, + cutoffMs: args.cutoffMs ?? null + }) +} + +it('reads a path the index holds nothing for, and lets the reader continue where it can', () => { + // Not `whole`: there is no span this index has to reach past, and the first + // enablement inside a running app has a warm list cursor to make use of. + expect(decide({ row: undefined })).toBe('any') +}) + +it('skips a file the index already covers at this stat', () => { + expect(decide({})).toBe('skip') +}) + +it('reads a file whose stat moved, however it moved', () => { + expect(decide({ file: { mtimeMs: MTIME + 1 } })).toBe('any') + // Grown without its mtime moving: a same-second append, or a restored stamp. + expect(decide({ file: { sizeBytes: 200 } })).toBe('any') +}) + +it('reads a file outside the retention window not at all', () => { + expect(decide({ row: undefined, cutoffMs: MTIME + 1 })).toBe('skip') + // And retention wins over everything else that would have asked for a read. + expect(decide({ row: row({ state: 'due' }), cutoffMs: MTIME + 1 })).toBe('skip') +}) + +it('reads a row owed a whole read from the start', () => { + expect(decide({ row: row({ state: 'due' }) })).toBe('whole') +}) + +it('reads whole rather than appending onto a cursor that continues nothing', () => { + // A different file at the same name: the identity check hands back no cursor. + expect(decide({ cursor: null })).toBe('whole') + // A chunked read that committed a prefix and no offset any append continues. + expect(decide({ cursor: { byteOffset: null, mtimeMs: MTIME, sizeBytes: 100 } })).toBe('whole') + // Shorter than the index read to, so this is not that file any more. + expect(decide({ file: { sizeBytes: 40 }, cursor })).toBe('whole') +}) + +it('retries a failed read until it has failed enough times at one stat', () => { + for (let failures = 1; failures < SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT; failures++) { + expect( + decide({ row: row({ state: 'failed', failCount: failures, failedMtimeMs: MTIME }) }) + ).toBe('any') + } + expect( + decide({ + row: row({ + state: 'failed', + failCount: SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT, + failedMtimeMs: MTIME + }) + }) + ).toBe('skip') +}) + +it('starts trying again the moment a held-out file changes', () => { + // The stat is the whole release condition, so nothing has to remember when + // the failures happened or schedule a retry. + expect( + decide({ + file: { mtimeMs: MTIME + 1 }, + row: row({ state: 'failed', failCount: 9, failedMtimeMs: MTIME }) + }) + ).toBe('any') +}) diff --git a/src/main/ai-vault-search/session-search-read-decision.ts b/src/main/ai-vault-search/session-search-read-decision.ts new file mode 100644 index 00000000000..cb25ad27ccb --- /dev/null +++ b/src/main/ai-vault-search/session-search-read-decision.ts @@ -0,0 +1,100 @@ +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type { SessionParseReadRequirement } from '../ai-vault/session-scanner-parse-cache' +import { requiresWholeRead, type SessionSearchIndexedFile } from './session-search-file-cursor' +import type { SessionSearchFileRow } from './session-search-store' + +/** + * Failures at one unchanged stat before a file is left alone. + * + * Three rather than one, because a single failure is often a transcript being + * rewritten under the read; three at the same mtime is not. The retry policy is + * the stat itself: an edit, a restore, or a `touch` after a `chmod` all move it, + * and nothing else does, so no timer is needed and none is kept. + */ +export const SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT = 3 + +/** + * What a pass owes one candidate: nothing, a read, or a read from the start. + * + * `any` and `whole` are the reader's own lanes. `whole` drops the session + * list's resume point, which is the only way to reach a span this index never + * saw; `any` asks for some bytes and lets the reader continue where it can, + * which is what the first enablement inside a running app needs — a warm list + * cursor sitting at the file's current stat would otherwise open nothing. + */ +export type SessionSearchReadDecision = 'skip' | SessionParseReadRequirement + +/** + * The whole of the indexer's decide step, as a function of the candidate's stat + * and the row the store holds for it. No pass state, no queue, no memory: the + * same inputs give the same answer on the first pass after a restart as on the + * hundredth of a long-running process, which is what lets a deadline cut a pass + * short with nothing to record. What did not get read is still owed, because + * being owed is a fact about the row. + */ +export function sessionSearchReadDecision(args: { + candidate: SessionFileCandidate + /** The file table's row, or undefined when the index holds nothing for it. */ + row: SessionSearchFileRow | undefined + /** The cursor for this candidate's identity; null when it is not continuable. */ + cursor: SessionSearchIndexedFile | null + /** Oldest transcript mtime worth holding rows for, or null for all history. */ + cutoffMs: number | null +}): SessionSearchReadDecision { + const { candidate, row, cursor, cutoffMs } = args + const file = candidate.file + // Retention first: a file outside the window is not worth reading whatever + // else is true of it, and the purge is what removes any row it still has. + if (cutoffMs !== null && file.mtimeMs < cutoffMs) { + return 'skip' + } + if (!row) { + // Nothing held for this path. Not `whole`, because the reader can continue + // from wherever it likes: there is no span this index has to reach past. + return 'any' + } + if (heldOut(row, file.mtimeMs)) { + return 'skip' + } + if (row.state === 'due') { + // The index is behind on a span no append reaches: a declined append, or a + // window that widened to admit this file. + return 'whole' + } + if (cursor === null || requiresWholeRead(cursor)) { + // A different file at the same name, or a chunked read that left a prefix + // and no cursor. Appending onto either would splice two spans together. + return 'whole' + } + const size = file.sizeBytes + if (typeof size === 'number' && cursor.byteOffset !== null && cursor.byteOffset > size) { + // Shorter than the index read to: this is not the file that cursor came from. + return 'whole' + } + if (row.state === 'failed') { + // Still within its retries, or the stat moved since it last failed. + return 'any' + } + return statMatches(row, file) ? 'skip' : 'any' +} + +/** + * True when this file has failed enough times at exactly this stat to stop + * trying. The stat is the whole release condition, so a file nobody touches is + * never read again and one that changes is read on the next pass that sees it. + */ +function heldOut(row: SessionSearchFileRow, mtimeMs: number): boolean { + return ( + row.state === 'failed' && + row.failCount >= SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT && + row.failedMtimeMs === mtimeMs + ) +} + +/** The row already describes the file as it is now. */ +function statMatches(row: SessionSearchFileRow, file: SessionFileCandidate['file']): boolean { + return ( + row.mtimeMs === file.mtimeMs && + (row.sizeBytes === null || file.sizeBytes === undefined || row.sizeBytes === file.sizeBytes) + ) +} diff --git a/src/main/ai-vault-search/session-search-retention-policy.test.ts b/src/main/ai-vault-search/session-search-retention-policy.test.ts new file mode 100644 index 00000000000..71c6cdb0862 --- /dev/null +++ b/src/main/ai-vault-search/session-search-retention-policy.test.ts @@ -0,0 +1,26 @@ +import { expect, it } from 'vitest' +import { sessionSearchHistoryCutoffMs } from './session-search-retention-policy' + +const NOW = 1_740_000_000_000 + +it('treats a fractional or non-positive day count as no bound at all', () => { + // A day count that floors to zero would read as "all history" in one place + // and "cutoff is now" in the other; both sides answer null. + expect(sessionSearchHistoryCutoffMs(0.4, NOW)).toBeNull() + expect(sessionSearchHistoryCutoffMs(0, NOW)).toBeNull() + expect(sessionSearchHistoryCutoffMs(-30, NOW)).toBeNull() + expect(sessionSearchHistoryCutoffMs(30, NOW)).toBe(NOW - 30 * 86_400_000) + // Clamped rather than unbounded: a caller asking for three thousand years of + // history gets the ceiling, not an mtime before the epoch. + expect(sessionSearchHistoryCutoffMs(999_999, NOW)).toBe(NOW - 3_650 * 86_400_000) +}) + +// The cutoff is read from the clock on every pass, not frozen at construction: +// a purge and the accept check that follows it must not disagree about where +// the window is, or the sweep deletes rows the next candidate re-indexes. +it('moves the cutoff with the clock', () => { + const later = NOW + 86_400_000 + expect(sessionSearchHistoryCutoffMs(30, later)).toBe( + (sessionSearchHistoryCutoffMs(30, NOW) ?? 0) + 86_400_000 + ) +}) diff --git a/src/main/ai-vault-search/session-search-retention-policy.ts b/src/main/ai-vault-search/session-search-retention-policy.ts new file mode 100644 index 00000000000..c7fa8a0b6d1 --- /dev/null +++ b/src/main/ai-vault-search/session-search-retention-policy.ts @@ -0,0 +1,25 @@ +const DAY_MS = 86_400_000 +const HISTORY_DAYS_MAX = 3_650 + +/** + * The retention window, as the indexer's callers state it and as the store + * consumes it. Settings storage is PR 3b's problem; this is the arithmetic. + */ +function normalizeSessionSearchHistoryDays(value: number | null): number | null { + if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) { + return null + } + // Why floor then re-check: a fractional day floors to 0, which reads as "all + // history" on one side and "now" on the other; make the two agree. + const days = Math.floor(value) + return days <= 0 ? null : Math.min(HISTORY_DAYS_MAX, days) +} + +/** The oldest transcript mtime worth indexing; null means no bound. */ +export function sessionSearchHistoryCutoffMs( + historyDays: number | null, + nowMs: number +): number | null { + const days = normalizeSessionSearchHistoryDays(historyDays) + return days === null ? null : nowMs - days * DAY_MS +} diff --git a/src/main/ai-vault-search/session-search-scan-roots.test.ts b/src/main/ai-vault-search/session-search-scan-roots.test.ts new file mode 100644 index 00000000000..51b7381c78b --- /dev/null +++ b/src/main/ai-vault-search/session-search-scan-roots.test.ts @@ -0,0 +1,58 @@ +import { expect, it } from 'vitest' +import { delimiter, join } from 'node:path' +import type { SessionFileDiscovery } from '../ai-vault/session-scanner-types' +import { sessionSearchRootListings } from './session-search-scan-roots' + +const STATE = '/tmp/ss-roots/openclaw-state' +const LEGACY = '/tmp/ss-roots/openclaw-legacy' + +function file(path: string): SessionFileDiscovery['files'][number] { + return { path, mtimeMs: 0, modifiedAt: new Date(0).toISOString() } +} + +it('splits a merged discovery into the real directories behind it', () => { + const current = join(STATE, 'agents') + const legacy = join(LEGACY, 'agents') + const listings = sessionSearchRootListings( + { openclawStateDir: STATE, openclawLegacyStateDir: LEGACY }, + [ + { + agent: 'openclaw', + // What discovery reports for an agent whose roots are alternates. + rootDir: [current, legacy].join(delimiter), + files: [ + file(join(current, 'a', 'sessions', 'one.jsonl')), + file(join(current, 'a', 'sessions', 'two.jsonl')), + file(join(legacy, 'b', 'sessions', 'three.jsonl')) + ] + } + ] + ) + + const byRoot = Object.fromEntries(listings.map((one) => [one.root, one.files])) + expect(byRoot[current]).toBe(2) + expect(byRoot[legacy]).toBe(1) + // The joined string is never reported as a directory. + expect(listings.every((one) => !one.root.includes(delimiter))).toBe(true) +}) + +it('attributes a file by path segment, not by string prefix', () => { + const agents = join(STATE, 'agents') + const legacy = join(LEGACY, 'agents') + const listings = sessionSearchRootListings( + { openclawStateDir: STATE, openclawLegacyStateDir: LEGACY }, + [ + { + agent: 'openclaw', + rootDir: [agents, legacy].join(delimiter), + // A sibling directory whose name merely starts with a root's name. It + // is under no root, so it belongs to none of them. + files: [file(join(`${agents}-old`, 'b', 'sessions', 'two.jsonl'))] + } + ] + ) + + const byRoot = Object.fromEntries(listings.map((one) => [one.root, one.files])) + expect(byRoot[agents]).toBe(0) + expect(byRoot[legacy]).toBe(0) +}) diff --git a/src/main/ai-vault-search/session-search-scan-roots.ts b/src/main/ai-vault-search/session-search-scan-roots.ts new file mode 100644 index 00000000000..8df3510199e --- /dev/null +++ b/src/main/ai-vault-search/session-search-scan-roots.ts @@ -0,0 +1,135 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import { AI_VAULT_AGENT_SOURCES } from '../ai-vault/session-scanner-agent-sources' +import { normalizedWslHomeDirs } from '../ai-vault/session-scanner-roots' +import { sessionCandidatesFromDiscoveries } from '../ai-vault/session-scanner-candidates' +import { discoverAiVaultSessionSources } from '../ai-vault/session-scanner-source-discovery' +import type { + AiVaultScanOptions, + SessionFileCandidate, + SessionFileDiscovery +} from '../ai-vault/session-scanner-types' + +/** One real directory a scan walked, and what it listed there. */ +export type SessionSearchRootListing = { root: string; files: number } + +/** + * Where the indexer looks. The caller resolves these so the index enumerates + * exactly the trees the session list does; the indexer owns the bounds + * (`limit`, `limitPerAgent`, `unlimited`) and its own cancellation, so those + * are not the caller's to set. + */ +export type SessionSearchScanRoots = Omit< + AiVaultScanOptions, + 'signal' | 'limit' | 'unlimited' | 'limitPerAgent' | 'scopePaths' +> + +export type SessionSearchDiscovery = { + /** Newest first, Codex hardlink aliases collapsed, exactly as a list scan sees them. */ + candidates: SessionFileCandidate[] + discoveries: SessionFileDiscovery[] + issues: AiVaultScanIssue[] +} + +/** + * The discovery half of a list scan, without the parse. `limitPerAgent` is the + * sidebar's own recency rule (`SessionNewestFiles` keeps the newest N per root); + * passing Infinity is what makes a sweep whole. + */ +export async function discoverSessionSearchCandidates( + roots: SessionSearchScanRoots, + args: { limitPerAgent: number; signal?: AbortSignal } +): Promise { + const issues: AiVaultScanIssue[] = [] + const options: AiVaultScanOptions = { ...roots, signal: args.signal } + const discoveries = await discoverAiVaultSessionSources({ + options, + limitPerAgent: args.limitPerAgent, + issues + }) + const candidates = await sessionCandidatesFromDiscoveries(discoveries, options) + return { candidates, discoveries, issues } +} + +/** + * Containment on path segments, not on string prefix, and on both separators: + * discovery joins with the platform's, a configured root can arrive spelled + * with the other, and `/a/agents-old` is not inside `/a/agents`. + */ +export function isUnderScanRoot(path: string, root: string): boolean { + return root.length > 0 && (path.startsWith(`${root}/`) || path.startsWith(`${root}\\`)) +} + +/** + * The real directories behind a scan's discoveries, with their file counts. + * + * Why this exists: an agent whose roots are alternates for one install reports + * them as a single discovery whose `rootDir` is every path joined by the + * platform's path delimiter. That string is not a directory. Health probes + * readdir it and get ENOENT, a containment check never matches a file under it, + * and a scan issue recorded against a real root never equals it — so the fence + * meant to protect an unmounted tree is inert for exactly the agent most likely + * to have one. Splitting the joined string back apart would be worse: a + * directory may legally contain the delimiter. The constituent paths come from + * the same source table discovery read. + */ +export function sessionSearchRootListings( + roots: SessionSearchScanRoots, + discoveries: readonly SessionFileDiscovery[] +): SessionSearchRootListing[] { + const wslHomeDirs = normalizedWslHomeDirs(roots.wslHomeDirs) + const counts = new Map() + for (const discovery of discoveries) { + const constituents = constituentRoots(roots, wslHomeDirs, discovery) + for (const root of constituents) { + counts.set(root, counts.get(root) ?? 0) + } + for (const file of discovery.files) { + const owner = owningRoot(constituents, file.path) + if (owner !== null) { + counts.set(owner, (counts.get(owner) ?? 0) + 1) + } + } + } + return [...counts].map(([root, files]) => ({ root, files })) +} + +function constituentRoots( + roots: SessionSearchScanRoots, + wslHomeDirs: readonly string[], + discovery: SessionFileDiscovery +): string[] { + const declared = AI_VAULT_AGENT_SOURCES[discovery.agent]?.rootDirs(roots, wslHomeDirs) ?? [] + if (declared.includes(discovery.rootDir)) { + return [discovery.rootDir] + } + // Either a merged discovery, whose rootDir is the joined string, or a source + // that builds its own discoveries (OpenCode, Antigravity) and reports a real + // directory that this table does not list. + return declared.length > 0 ? declared : [discovery.rootDir] +} + +function owningRoot(constituents: readonly string[], path: string): string | null { + let owner: string | null = null + for (const root of constituents) { + if (isUnderScanRoot(path, root) && (owner === null || root.length > owner.length)) { + owner = root + } + } + return owner +} + +/** + * Roots that listed transcripts on the previous pass and list none on this one. + * + * The one bit of memory the retirement walk gets, and what it buys: a root that + * blinks empty for a single pass is unverifiable rather than proven gone, so a + * sync client swapping a directory out cannot retire a tree. It is deliberately + * not evidence that survives the process — see the invariant block in + * `session-search-deleted-sources.ts` for what that costs and why. + */ +export function sessionSearchEmptiedRoots( + previous: ReadonlySet, + current: ReadonlySet +): Set { + return new Set([...previous].filter((root) => !current.has(root))) +} diff --git a/src/main/ai-vault-search/session-search-schema.ts b/src/main/ai-vault-search/session-search-schema.ts index 2da1e64bc11..ca525c47c0e 100644 --- a/src/main/ai-vault-search/session-search-schema.ts +++ b/src/main/ai-vault-search/session-search-schema.ts @@ -57,7 +57,18 @@ CREATE TABLE IF NOT EXISTS files( byte_offset INTEGER NOT NULL, mtime_ms REAL NOT NULL, size_bytes INTEGER, - session_row_id INTEGER + session_row_id INTEGER, + -- What this row still owes a reader, so that nothing has to be remembered + -- between passes. 'current': the rows match the file at the stat recorded + -- here. 'due': the index is behind on content it cannot reach by appending, + -- so the next pass reads the file whole. 'failed': the last read did not + -- commit, and the two columns below are what stop it being retried for ever. + state TEXT NOT NULL DEFAULT 'current', + fail_count INTEGER NOT NULL DEFAULT 0, + -- The mtime the failures were observed at. A file that fails at one stat is + -- left alone once it has failed enough times, and only a change to this stat + -- can mean the file itself changed, so it is the whole retry policy. + failed_mtime_ms REAL ); -- Retention walks the expiring end of this column; without it that is a full scan and a sort. CREATE INDEX IF NOT EXISTS files_mtime ON files(mtime_ms); diff --git a/src/main/ai-vault-search/session-search-store-is-memory.test.ts b/src/main/ai-vault-search/session-search-store-is-memory.test.ts new file mode 100644 index 00000000000..2e90a75862f --- /dev/null +++ b/src/main/ai-vault-search/session-search-store-is-memory.test.ts @@ -0,0 +1,265 @@ +import { chmod, rm, utimes } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +/* + * S1-S5: the store is the only memory. + * + * Every question the indexer answers between passes -- what is owed a read, + * what has failed and how often, what it holds and therefore what may have been + * deleted, what to report -- is a row in the `files` table. These tests check + * that from outside the object: a second connection, hand-written SQL, and the + * clock. Two things outlive a pass and are not rows, and both are named here: + * the timer, and one bit per root for the retirement walk's grace. + */ + +const INTERVAL_MS = 20_000 +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const FIRST = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +const SECOND = 'bbbbbbbb-cccc-4ddd-8eee-ffffffffffff' +const THIRD = 'cccccccc-dddd-4eee-8fff-000000000000' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-memory') + indexer = null +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function newIndexer( + overrides: Partial[0]> = {} +): SessionSearchIndexer { + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS, + onError: (error) => errors.push(error), + ...overrides + }) + return indexer +} + +function transcriptPath(name: string): string { + return join(harness.claudeProjectDir, `${name}.jsonl`) +} + +async function nextCycle(): Promise { + clock.advance(INTERVAL_MS) + await indexer?.settled() +} + +/** The whole `files` table as a second connection sees it, ordered for comparison. */ +function fileTable(): unknown[] { + return harness.read((db: SyncDatabase) => + db + .prepare( + `SELECT path, dev, ino, byte_offset, mtime_ms, size_bytes, session_row_id, + state, fail_count, failed_mtime_ms + FROM files ORDER BY path` + ) + .all() + ) +} + +// S1. The status is a query. A counter kept beside the rows is what needs a +// rule about when to reset, and every such rule this feature grew was wrong. +it('S1: reports exactly what a hand-written query over the rows reports', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['one'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['two'], SECOND) + await newIndexer().start() + + const bySql = (): Record => + Object.fromEntries( + ( + harness.read((db: SyncDatabase) => + db.prepare('SELECT state, count(*) AS n FROM files GROUP BY state').all() + ) as { state: string; n: number }[] + ).map((row) => [row.state, Number(row.n)]) + ) + + const reported = indexer?.status() + const counted = bySql() + expect(reported?.filesIndexed).toBe(counted.current ?? 0) + expect(reported?.filesDue).toBe(counted.due ?? 0) + expect(reported?.filesFailed).toBe(counted.failed ?? 0) + expect(reported?.filesIndexed).toBe(2) + + // And it stays a query: delete a row behind the indexer's back and the very + // next call reports the table, not a number it remembered. + harness.write((db: SyncDatabase) => + db.prepare('DELETE FROM files WHERE path = ?').run(transcriptPath(FIRST)) + ) + expect(indexer?.status().filesIndexed).toBe(1) +}) + +// S2. A deletion is proven by comparing the rows against what discovery +// returned, so the moment it happened does not matter. Every boundary a pass +// has is a moment a file can go. +it('S2: retires a file deleted right after the opening sweep', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await newIndexer().start() + + await rm(transcriptPath(FIRST)) + await nextCycle() + + expect(fileTable()).toHaveLength(1) +}) + +it('S2: retires a file deleted right after a cycle', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await newIndexer().start() + await nextCycle() + + await rm(transcriptPath(FIRST)) + await nextCycle() + + expect(fileTable()).toHaveLength(1) +}) + +it('S2: retires a file deleted right after a periodic sweep', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await newIndexer({ fullSweepEveryCycles: 2 }).start() + await nextCycle() + await nextCycle() + // The third pass is the periodic sweep; the file goes the moment it ends. + await nextCycle() + + await rm(transcriptPath(FIRST)) + await nextCycle() + + expect(fileTable()).toHaveLength(1) +}) + +it('S2: retires a file deleted while a pass was out of time', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await writeClaudeTranscript(transcriptPath(THIRD), ['also staying'], THIRD) + // One transcript a pass: the opening sweep leaves two of the three unread. + clock.costPerNowMs = 1_000 + await newIndexer({ passDeadlineMs: 1_000 }).start() + expect(fileTable()).toHaveLength(1) + + await rm(transcriptPath(FIRST)) + await nextCycle() + await nextCycle() + + // Read what it could, and proved the deletion in the same pass it was still + // catching up in: retirement is not what the deadline bounds. + expect((fileTable() as { path: string }[]).map((row) => row.path)).not.toContain( + transcriptPath(FIRST) + ) +}) + +// S3. The stat is the whole retry policy: a file that fails at one stat stops +// being read, and only a change to that stat starts it again. +it.skipIf(!CAN_DENY_READ)( + 'S3: stops reading a file that fails three times at one stat', + async () => { + const path = transcriptPath(FIRST) + await writeClaudeTranscript(path, ['behind the wrong mode bits'], FIRST) + await chmod(path, 0o000) + try { + await newIndexer().start() + for (let cycle = 0; cycle < 4; cycle++) { + await nextCycle() + } + + const row = harness.read((db: SyncDatabase) => + db.prepare('SELECT state, fail_count AS failCount FROM files WHERE path = ?').get(path) + ) as { state: string; failCount: number } + // Three, not four and not seven: the pass after the third costs nothing. + expect(row).toEqual({ state: 'failed', failCount: 3 }) + expect(indexer?.status()).toMatchObject({ filesFailed: 1, phase: 'degraded' }) + + // Only the stat releases it. + await chmod(path, 0o644) + const later = new Date(Date.now() + 60_000) + await utimes(path, later, later) + await nextCycle() + + expect(indexer?.status()).toMatchObject({ filesIndexed: 1, filesFailed: 0 }) + } finally { + await chmod(path, 0o644) + } + } +) + +// S4. Two passes over an unchanged filesystem leave the table byte for byte as +// they found it. Anything that drifted would be state the rows do not hold. +it('S4: leaves the file table identical across passes with no change on disk', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['one'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['two'], SECOND) + await newIndexer({ fullSweepEveryCycles: 2 }).start() + + const afterSweep = fileTable() + await nextCycle() + expect(fileTable()).toEqual(afterSweep) + await nextCycle() + expect(fileTable()).toEqual(afterSweep) + // Including across the periodic sweep, which reads the same rows again. + await nextCycle() + expect(fileTable()).toEqual(afterSweep) + expect(errors).toEqual([]) +}) + +// S5. Nothing a close interrupts needs repairing: the next instance reads the +// rows as they stand and decides from them alone. +it('S5: leaves the store consistent when a close interrupts a pass', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['one'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['two'], SECOND) + await writeClaudeTranscript(transcriptPath(THIRD), ['three'], THIRD) + newIndexer() + let closed = false + clock.onNow = () => { + if (closed || fileTable().length === 0) { + return + } + closed = true + indexer?.close() + } + await indexer?.start() + await indexer?.settled() + clock.onNow = null + + const interrupted = fileTable() + expect(interrupted.length).toBeGreaterThan(0) + expect(interrupted.length).toBeLessThan(3) + expect(errors).toEqual([]) + + // A new instance over the same database: no repair pass, no recovery, just + // the rows and what they say is owed. + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await newIndexer().start() + + expect(indexer?.status()).toMatchObject({ filesIndexed: 3, filesDue: 0, filesFailed: 0 }) +}) diff --git a/src/main/ai-vault-search/session-search-store.ts b/src/main/ai-vault-search/session-search-store.ts index da9f4d6f731..daa9267561b 100644 --- a/src/main/ai-vault-search/session-search-store.ts +++ b/src/main/ai-vault-search/session-search-store.ts @@ -13,10 +13,36 @@ import { import { deleteExpiredSearchFiles, drainOrphanedMessages } from './session-search-retention-delete' import { openSessionSearchDatabase } from './session-search-schema' -// A paused store keeps recording what it declined, so the set needs a ceiling. -// Above it the oldest record goes and the drop is counted, because a re-read set -// that silently forgets is worse than one that says it is incomplete. -export const STALE_PATH_LIMIT = 20_000 +/** + * What a row still owes a reader. + * + * `current`: the rows match the file at the stat this row records. + * `due`: the index is behind on a span it cannot reach by appending, so the + * next pass must read the file whole. + * `failed`: the last read did not commit; `failCount` and `failedMtimeMs` are + * what stop it being retried for ever. + */ +export type SessionSearchFileState = 'current' | 'due' | 'failed' + +/** + * One row of the index's own file table. + * + * This is the indexer's whole memory between passes: what it holds, at what + * stat, and what each row still owes. Nothing it decides is answered from + * anywhere else, which is why a second connection can check its status. + */ +export type SessionSearchFileRow = { + path: string + identity: SessionSearchFileIdentity + mtimeMs: number + sizeBytes: number | null + state: SessionSearchFileState + failCount: number + failedMtimeMs: number | null +} + +/** How many rows are in each state; the whole of the indexer's progress report. */ +export type SessionSearchStateCounts = { current: number; due: number; failed: number } /** * Owns the index database. PR 2 scope: the write half only — the transcript @@ -27,12 +53,7 @@ export class SessionSearchStore { private readonly db: SyncDatabase private readonly writer: SessionSearchIndexWriter private closed = false - private acceptingWrites = true private retentionCutoffMs: number | null = null - // Files this index knows it is behind on. Filled by a declined or abandoned - // read; PR 3's indexer drains it. Nothing here schedules the re-read. - private readonly stale = new Map() - private droppedStalePaths = 0 // One drain at a time. A replace that commits while one is running asks for // another pass rather than starting a second walk of the same rows. private draining = false @@ -104,23 +125,26 @@ export class SessionSearchStore { return this.db } - setAcceptingWrites(accept: boolean): void { - this.acceptingWrites = accept - } - /** The oldest transcript mtime worth indexing; PR 3 derives it from the retention setting. */ setRetentionCutoffMs(cutoffMs: number | null): void { this.retentionCutoffMs = cutoffMs } - /** Whether this candidate is new enough to be worth holding rows for at all. */ - private withinRetention(candidate: SessionFileCandidate): boolean { - return this.retentionCutoffMs === null || candidate.file.mtimeMs >= this.retentionCutoffMs + /** The cutoff a caller's own decide step compares a candidate's mtime against. */ + get retentionCutoff(): number | null { + return this.retentionCutoffMs } - /** Whether a write for this candidate may start right now. */ - acceptsCandidate(candidate: SessionFileCandidate): boolean { - return !this.closed && this.acceptingWrites && this.withinRetention(candidate) + /** + * Whether this candidate is new enough to hold rows for. + * + * Enforced here as well as in the indexer's decide step, and not only there: + * the consumer observes every read the session list makes, not only the ones + * the index asked for, so a sidebar scan of a transcript outside the window + * would otherwise index rows the next purge deletes again. + */ + private withinRetention(candidate: SessionFileCandidate): boolean { + return this.retentionCutoffMs === null || candidate.file.mtimeMs >= this.retentionCutoffMs } indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null { @@ -139,7 +163,7 @@ export class SessionSearchStore { previousByteOffset: number, identity?: () => TranscriptSessionIdentity | null ): SessionSearchFileWrite | null { - if (!this.acceptsCandidate(candidate)) { + if (this.closed || !this.withinRetention(candidate)) { return null } try { @@ -150,10 +174,14 @@ export class SessionSearchStore { } } + /** + * A read that landed. Written after the commit rather than inside it: the + * transaction owns the rows and the cursor, and a crash between the two + * leaves a row that says `failed` over content that is in fact current, which + * the next pass fixes by reading a file it did not have to. + */ writeCommitted(candidate: SessionFileCandidate): void { - // Why: a list scan queues every file the backfill has not reached yet; once - // one lands, a later pass must not re-read the whole queue. - this.stale.delete(candidate.file.path) + this.setFileState(candidate.file.path, 'current') } reportWriteFailure(error: unknown): void { @@ -161,55 +189,89 @@ export class SessionSearchStore { } /** - * Records a file whose content the index is behind on, for a later whole - * re-read. Recorded while paused too: a pause is exactly the window in which - * reads are declined, so refusing to remember them would lose every file the - * pause covered. - */ - markStale(candidate: SessionFileCandidate): void { - if (this.closed || !this.withinRetention(candidate)) { - return - } - // Re-inserting moves the path to the end, so the oldest record is the one - // dropped when a long pause overruns the bound. - this.stale.delete(candidate.file.path) - this.stale.set(candidate.file.path, candidate) - while (this.stale.size > STALE_PATH_LIMIT) { - const oldest = this.stale.keys().next() - if (oldest.done) { - break - } - this.stale.delete(oldest.value) - this.droppedStalePaths += 1 - } - } - - /** - * Files the index knew it was behind on and could not keep a record of. A - * non-zero count means the re-read set is incomplete, so coverage cannot be - * reported as whole until a full pass runs. - */ - get droppedPendingFileCount(): number { - return this.droppedStalePaths - } - - /** - * Hands the re-read set to its scheduler and clears it. + * Every row this index holds. The candidate list for retirement and the whole + * of the status, read in one query so that no pass has to carry either. * - * These paths are behind, not merely dirty: the index declined their last read - * because it covered a span the index never saw. Re-dispatching a scan is not - * enough on its own, because the reader picks `append` from the session list's - * resume point and the consumer will decline again. The caller must pass each - * path to `requestWholeTranscriptRead` first. + * The cursor is deliberately not here: whether a row can be continued is + * `indexedFile`'s question, and one spelling of the half-written sentinel is + * enough. */ - takeStale(): SessionFileCandidate[] { - const candidates = [...this.stale.values()] - this.stale.clear() - return candidates + files(): SessionSearchFileRow[] { + return ( + this.db + .prepare( + `SELECT path, dev, ino, mtime_ms AS mtimeMs, size_bytes AS sizeBytes, + state, fail_count AS failCount, failed_mtime_ms AS failedMtimeMs + FROM files` + ) + .all() as (Omit & { + dev: number | null + ino: number | null + })[] + ).map((row) => ({ + path: row.path, + identity: + typeof row.dev === 'number' && typeof row.ino === 'number' + ? { dev: row.dev, ino: row.ino } + : null, + mtimeMs: row.mtimeMs, + sizeBytes: row.sizeBytes, + state: row.state, + failCount: row.failCount, + failedMtimeMs: row.failedMtimeMs + })) } - get pendingFileCount(): number { - return this.stale.size + /** + * Moves a row's read state. + * + * `failed` also counts the failure and records the stat it happened at, which + * is what lets the next pass tell "this file has never worked" from "this + * file has changed since it last failed". A path with no row is a no-op: the + * next pass reads it because the index holds nothing for it. + */ + setFileState(path: string, state: SessionSearchFileState, atMtimeMs?: number): void { + try { + if (state === 'failed') { + // Inserted when there is no row, because the common unreadable file is + // one the index never managed to hold: a transcript behind the wrong + // mode bits fails on its very first read, and with nowhere to write the + // count it would be read again on every pass for the life of the + // process. The cursor is zero and there is no session, which is what + // "the index holds nothing for this file" already looks like. + this.db + .prepare( + `INSERT INTO files(path, byte_offset, mtime_ms, state, fail_count, failed_mtime_ms) + VALUES (?, 0, ?, 'failed', 1, ?) + ON CONFLICT(path) DO UPDATE SET + state = 'failed', + fail_count = files.fail_count + 1, + failed_mtime_ms = excluded.failed_mtime_ms` + ) + .run(path, atMtimeMs ?? 0, atMtimeMs ?? null) + return + } + this.db + .prepare( + 'UPDATE files SET state = ?, fail_count = 0, failed_mtime_ms = NULL WHERE path = ?' + ) + .run(state, path) + } catch (error) { + this.onError(error) + } + } + + /** Rows per state. The status is this query and the pass's own degraded roots. */ + stateCounts(): SessionSearchStateCounts { + const rows = this.db.prepare('SELECT state, count(*) AS n FROM files GROUP BY state').all() as { + state: SessionSearchFileState + n: number + }[] + const counts: SessionSearchStateCounts = { current: 0, due: 0, failed: 0 } + for (const row of rows) { + counts[row.state] = Number(row.n) + } + return counts } /** @@ -218,7 +280,6 @@ export class SessionSearchStore { * (docs/reference/ssh-execution-boundary.md). */ removeFile(path: string): void { - this.stale.delete(path) try { this.writer.removeFile(path) } catch (error) { diff --git a/src/main/ai-vault-search/session-search-synthetic-sources.ts b/src/main/ai-vault-search/session-search-synthetic-sources.ts new file mode 100644 index 00000000000..86222b9c5fe --- /dev/null +++ b/src/main/ai-vault-search/session-search-synthetic-sources.ts @@ -0,0 +1,56 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import { splitOpenCodeSqliteCandidate } from '../ai-vault/session-scanner-opencode-sqlite-paths' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' + +/** + * A row whose path names a container and an entry inside it rather than a file + * of its own. OpenCode's SQLite sessions are the one shape today + * (`#`), which is why this reads through that source's + * own splitter rather than reinventing the encoding. + */ +export type SessionSearchSyntheticSource = { container: string; id: string } + +export function splitSyntheticSessionSource(path: string): SessionSearchSyntheticSource | null { + const openCode = splitOpenCodeSqliteCandidate(path) + return openCode ? { container: openCode.dbPath, id: openCode.sessionId } : null +} + +/** + * Which containers a pass enumerated in full, and every id each of them held. + * + * This is the synthetic equivalent of a directory listing, and it has to meet + * the same bar before the retirement walk may prove anything from it: + * + * - **Exhaustive.** Only a sweep enumerates without a per-agent limit. A cycle + * asks for the newest N, so an id it did not return may simply be the N+1th. + * Callers that are not a census do not build this at all. + * - **Successful.** A container a scan issue names could not be read, and a + * read that failed returns no ids rather than an error the walk can see. A + * named container is left out, so its rows stay unverifiable. + * - **Non-empty.** A container that returned nothing is not evidence that it + * holds nothing: a database whose schema this scanner no longer recognises + * returns an empty list with no error at all, and believing it would retire + * every session in one pass. The cost is one stale row per container whose + * last entry the user deletes, until the container gains an entry or goes. + */ +export function sessionSearchEnumeratedContainers( + candidates: readonly SessionFileCandidate[], + issues: readonly AiVaultScanIssue[] +): Map> { + const containers = new Map>() + for (const candidate of candidates) { + const synthetic = splitSyntheticSessionSource(candidate.file.path) + if (!synthetic) { + continue + } + const ids = containers.get(synthetic.container) ?? new Set() + ids.add(synthetic.id) + containers.set(synthetic.container, ids) + } + for (const issue of issues) { + if (issue.kind !== 'notice') { + containers.delete(issue.path) + } + } + return containers +} diff --git a/src/main/ai-vault-search/session-search-work-loop.ts b/src/main/ai-vault-search/session-search-work-loop.ts new file mode 100644 index 00000000000..41a0ad98497 --- /dev/null +++ b/src/main/ai-vault-search/session-search-work-loop.ts @@ -0,0 +1,87 @@ +import type { SessionSearchClock, SessionSearchTimerHandle } from './session-search-clock' + +export type SessionSearchWorkLoopOptions = { + clock: SessionSearchClock + intervalMs: number + /** A task that threw for a reason other than its own abort. */ + onFailure: (error: unknown) => void +} + +/** + * Runs the indexer's passes one at a time, on an interval, until it is closed. + * + * Separate from the indexer because it is the part with no opinion about + * transcripts: a task chain that never overlaps itself, a timer that only ever + * has one pending tick, and a close that cancels both. Arming inside the chain + * rather than beside it is what makes `settled` mean "everything queued so far + * has finished, including the re-arm", which is what a fake-clock test needs. + */ +export class SessionSearchWorkLoop { + private timer: SessionSearchTimerHandle | null = null + private controller: AbortController | null = null + private chain: Promise = Promise.resolve() + private closed = false + + constructor(private readonly options: SessionSearchWorkLoopOptions) {} + + /** Everything queued so far. Never rejects: a task's failure is reported, not thrown. */ + get settled(): Promise { + return this.chain + } + + /** Queues `work` behind whatever is running, then re-arms the interval. */ + queue(work: (signal: AbortSignal) => Promise, tick: () => void): Promise { + const chained = this.chain + .then( + () => this.run(work), + () => this.run(work) + ) + .then(() => this.arm(tick)) + this.chain = chained + return chained + } + + /** + * Stops the timer, the task in flight and everything queued behind it. Nothing + * queued before this call may run afterwards: that is what lets the indexer + * close its store here and know no pass will reach for it. + */ + close(): void { + this.closed = true + if (this.timer !== null) { + this.options.clock.clearTimeout(this.timer) + this.timer = null + } + this.controller?.abort() + } + + private arm(tick: () => void): void { + if (this.closed || this.timer !== null) { + return + } + this.timer = this.options.clock.setTimeout(() => { + this.timer = null + tick() + }, this.options.intervalMs) + } + + private async run(work: (signal: AbortSignal) => Promise): Promise { + if (this.closed) { + return + } + const controller = new AbortController() + this.controller = controller + try { + await work(controller.signal) + } catch (error) { + // An aborted task is a close, never a failure. + if (!controller.signal.aborted) { + this.options.onFailure(error) + } + } finally { + if (this.controller === controller) { + this.controller = null + } + } + } +} diff --git a/src/main/ai-vault/session-scanner-parse-cache.ts b/src/main/ai-vault/session-scanner-parse-cache.ts index 46fb4754a32..6ebb1a76a2b 100644 --- a/src/main/ai-vault/session-scanner-parse-cache.ts +++ b/src/main/ai-vault/session-scanner-parse-cache.ts @@ -26,6 +26,7 @@ import { import { readResumableTranscript, readWholeTranscript, + requestWholeTranscriptRead, type TranscriptReadStats } from './session-transcript-reader' @@ -104,29 +105,67 @@ export function createSessionParseStats(): SessionParseStats { export async function parseAgentSessionFileCached( candidate: SessionFileCandidate, platform: NodeJS.Platform, - stats?: SessionParseStats + stats?: SessionParseStats, + requireRead?: SessionParseReadRequirement ): Promise { // The whole lookup-read-store sequence runs in the lane: a concurrent parse of // the same path shares this entry's resume point and its message channel. return inSessionParseFileLane(candidate.file.path, () => - parseCachedInLane(candidate, platform, stats) + parseCachedInLane(candidate, platform, stats, requireRead) + ) +} + +/** + * What a caller other than the session list needs out of this parse. + * + * `any`: some bytes must be read. A cursor already at the file's current stat + * is dropped so the reader opens it; one that is merely behind is left alone, + * because an append is a read. + * + * `whole`: the file must be re-read from zero, for a consumer whose own cursor + * covers a span this one does not. + * + * Why it is a parameter and not two calls around this one: the decision reads + * cache state and then changes it, so outside the per-path lane an overlapping + * list parse can store its entry in between and the forced read silently + * degrades to a reuse. + */ +export type SessionParseReadRequirement = 'any' | 'whole' + +/** + * True when this cursor already sits at the transcript's current stat, so a + * parse would reuse the cached fold and read no bytes at all. + */ +function sessionParseCacheCoversTranscript( + candidate: SessionFileCandidate, + platform: NodeJS.Platform +): boolean { + const { file } = candidate + const entry = getSessionParseCacheEntry(file.path) + return ( + entry !== undefined && + entry.platform === platform && + entry.mtimeMs === file.mtimeMs && + (entry.sizeBytes === null || file.sizeBytes === undefined || entry.sizeBytes === file.sizeBytes) ) } async function parseCachedInLane( candidate: SessionFileCandidate, platform: NodeJS.Platform, - stats?: SessionParseStats + stats?: SessionParseStats, + requireRead?: SessionParseReadRequirement ): Promise { const { file } = candidate + if ( + requireRead === 'whole' || + (requireRead === 'any' && sessionParseCacheCoversTranscript(candidate, platform)) + ) { + requestWholeTranscriptRead(file.path) + } const entry = getSessionParseCacheEntry(file.path) - const transcriptUnchanged = - entry !== undefined && - entry.platform === platform && - entry.mtimeMs === file.mtimeMs && - (entry.sizeBytes === null || file.sizeBytes === undefined || entry.sizeBytes === file.sizeBytes) - if (transcriptUnchanged) { + if (entry !== undefined && sessionParseCacheCoversTranscript(candidate, platform)) { if (sidecarUnchanged(entry.sidecar, file.sidecar)) { return reuseCachedSession(candidate, entry, stats) } From 2ecde717b4561cae1701a27615f704434232399a Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 11 Sep 2026 01:32:56 -0400 Subject: [PATCH 4/4] refactor(rpc): make defineMethod preserve the method name and handler result (#20016) --- .../push/push-registration-rpc.test.ts | 4 +- .../rpc/core-typed-method-contract.test.ts | 114 +++++++++++++++++ src/main/runtime/rpc/core.ts | 115 ++++++++++++------ .../runtime/rpc/dispatcher-request-parsing.ts | 4 +- src/main/runtime/rpc/dispatcher.ts | 7 +- src/main/runtime/rpc/methods/accounts.test.ts | 4 +- src/main/runtime/rpc/methods/accounts.ts | 4 +- .../runtime/rpc/methods/agent-hooks.test.ts | 4 +- src/main/runtime/rpc/methods/agent-hooks.ts | 4 +- src/main/runtime/rpc/methods/agent-session.ts | 4 +- src/main/runtime/rpc/methods/ai-vault.ts | 4 +- src/main/runtime/rpc/methods/artifacts.ts | 4 +- .../automation-scoped-list-methods.test.ts | 4 +- src/main/runtime/rpc/methods/automations.ts | 4 +- .../methods/browser-client-file-channel.ts | 4 +- .../rpc/methods/browser-client-host.ts | 4 +- src/main/runtime/rpc/methods/browser-core.ts | 4 +- .../runtime/rpc/methods/browser-extras.ts | 4 +- .../rpc/methods/browser-network-tunnel.ts | 4 +- .../runtime/rpc/methods/browser-screencast.ts | 4 +- .../rpc/methods/browser-text-rpc-methods.ts | 4 +- .../runtime/rpc/methods/client-events.test.ts | 9 +- src/main/runtime/rpc/methods/client-events.ts | 4 +- src/main/runtime/rpc/methods/client-ui.ts | 4 +- src/main/runtime/rpc/methods/clipboard.ts | 4 +- .../rpc/methods/computer-actions.test.ts | 3 +- src/main/runtime/rpc/methods/computer.test.ts | 4 +- src/main/runtime/rpc/methods/computer.ts | 4 +- src/main/runtime/rpc/methods/diagnostics.ts | 4 +- src/main/runtime/rpc/methods/emulator.ts | 4 +- .../rpc/methods/files-mutation-methods.ts | 4 +- .../files-terminal-artifact-methods.ts | 4 +- src/main/runtime/rpc/methods/files.ts | 4 +- .../runtime/rpc/methods/folder-workspace.ts | 4 +- .../git-commit-message-generation-methods.ts | 4 +- .../runtime/rpc/methods/git-diff-methods.ts | 4 +- src/main/runtime/rpc/methods/git.ts | 4 +- .../rpc/methods/github-issue-methods.ts | 4 +- .../rpc/methods/github-project-methods.ts | 4 +- .../methods/github-pull-request-methods.ts | 4 +- .../github-pull-request-update-methods.ts | 4 +- .../methods/github-repo-work-item-methods.ts | 4 +- src/main/runtime/rpc/methods/github.ts | 3 +- src/main/runtime/rpc/methods/gitlab.ts | 4 +- .../runtime/rpc/methods/host-capabilities.ts | 4 +- src/main/runtime/rpc/methods/hosted-review.ts | 4 +- src/main/runtime/rpc/methods/index.ts | 3 +- src/main/runtime/rpc/methods/jira.ts | 4 +- .../rpc/methods/linear-agent-access.ts | 4 +- .../rpc/methods/linear-project-create.ts | 4 +- src/main/runtime/rpc/methods/linear.ts | 4 +- .../methods/mobile-markdown-tab-methods.ts | 4 +- src/main/runtime/rpc/methods/native-chat.ts | 4 +- src/main/runtime/rpc/methods/notifications.ts | 4 +- src/main/runtime/rpc/methods/orchestration.ts | 3 +- .../federated-release-safety.test.ts | 5 +- .../federation/federation-control.ts | 4 +- .../federation-liveness-verdict.test.ts | 9 +- .../federation/federation-methods.ts | 3 +- .../federation/federation-relay.ts | 2 +- .../orchestration/federation/federation.ts | 4 +- .../rpc/methods/orchestration/gates/gates.ts | 4 +- .../orchestration/messaging/ask-methods.ts | 4 +- .../orchestration/messaging/check-methods.ts | 4 +- .../check-worker-federated-attachment.test.ts | 6 +- .../messaging/message-methods.ts | 4 +- .../orchestration/messaging/send-methods.ts | 4 +- .../methods/orchestration/rpc-test-harness.ts | 4 +- .../orchestration/runs/dispatch-methods.ts | 4 +- .../runs/mutation-request-show.ts | 4 +- .../orchestration/runs/reset-methods.ts | 4 +- .../rpc/methods/orchestration/runs/runs.ts | 4 +- .../manual-dispatch-observation.test.ts | 15 ++- .../worker/manual-dispatch-release.test.ts | 5 +- .../orchestration/worker/worker-control.ts | 4 +- .../worker/worker-list-method.ts | 4 +- .../orchestration/worker/worker-methods.ts | 3 +- .../worker-release-mobile-report.test.ts | 4 +- .../worker/worker-release-recovery.test.ts | 6 +- .../worker/worker-release.test-support.ts | 4 +- .../orchestration/worker/worker-release.ts | 4 +- .../worker-stop-liveness-verdict.test.ts | 5 +- .../orchestration/worker/worker-stop.ts | 4 +- .../worker/workers-recovery.test.ts | 5 +- .../methods/orchestration/worker/workers.ts | 4 +- src/main/runtime/rpc/methods/pairing.ts | 4 +- src/main/runtime/rpc/methods/plugins.test.ts | 4 +- src/main/runtime/rpc/methods/plugins.ts | 4 +- src/main/runtime/rpc/methods/preflight.ts | 4 +- .../methods/project-runtime-rpc-methods.ts | 4 +- src/main/runtime/rpc/methods/repo.ts | 4 +- .../methods/runtime-client-capabilities.ts | 4 +- .../rpc/methods/session-tab-close-methods.ts | 4 +- .../methods/session-tab-markdown-methods.ts | 4 +- .../methods/session-tab-mutation-methods.ts | 4 +- src/main/runtime/rpc/methods/session-tabs.ts | 4 +- src/main/runtime/rpc/methods/skills.test.ts | 8 +- src/main/runtime/rpc/methods/skills.ts | 4 +- src/main/runtime/rpc/methods/speech.ts | 4 +- src/main/runtime/rpc/methods/ssh.ts | 4 +- src/main/runtime/rpc/methods/stats.ts | 4 +- src/main/runtime/rpc/methods/status.ts | 4 +- .../methods/structured-agent-session-hold.ts | 4 +- .../structured-agent-session-reveal.ts | 4 +- .../structured-agent-session-status-stream.ts | 4 +- .../rpc/methods/structured-agent-session.ts | 4 +- .../structured-worker-stop-receipt.test.ts | 5 +- .../terminal-create-idempotency.test.ts | 14 ++- ...terminal-manifest-characterization.test.ts | 5 +- .../runtime/rpc/methods/terminal-orphan.ts | 4 +- src/main/runtime/rpc/methods/terminal.ts | 3 +- .../terminal-inspect-process-params.test.ts | 7 +- .../terminal/terminal-lifecycle-methods.ts | 4 +- .../terminal/terminal-multiplex-method.ts | 4 +- .../terminal/terminal-query-methods.ts | 4 +- .../methods/terminal/terminal-send-method.ts | 4 +- .../terminal/terminal-subscribe-method.ts | 4 +- .../terminal/terminal-viewport-methods.ts | 6 +- src/main/runtime/rpc/methods/updater.test.ts | 5 +- src/main/runtime/rpc/methods/updater.ts | 4 +- .../runtime/rpc/methods/workspace-ports.ts | 4 +- .../rpc/methods/worktree-catalog-methods.ts | 4 +- src/main/runtime/rpc/methods/worktree.ts | 4 +- .../runtime-rpc/runtime-rpc-pairing-types.ts | 4 +- 124 files changed, 482 insertions(+), 280 deletions(-) create mode 100644 src/main/runtime/rpc/core-typed-method-contract.test.ts diff --git a/src/main/runtime/push/push-registration-rpc.test.ts b/src/main/runtime/push/push-registration-rpc.test.ts index d01b561417d..78a91a238fa 100644 --- a/src/main/runtime/push/push-registration-rpc.test.ts +++ b/src/main/runtime/push/push-registration-rpc.test.ts @@ -2,14 +2,14 @@ import { mkdtempSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { describe, expect, it, vi } from 'vitest' -import type { RpcContext, RpcMethod } from '../rpc/core' +import { eraseRpcMethods, type RpcContext, type RpcMethod } from '../rpc/core' import { NOTIFICATION_METHODS } from '../rpc/methods/notifications' import { DeviceRegistry } from '../device-registry' import { OrcaRuntimeRpcServer } from '../runtime-rpc' import { OrcaRuntimeService } from '../orca-runtime' function method(name: string): RpcMethod { - const found = NOTIFICATION_METHODS.find((candidate) => candidate.name === name) + const found = eraseRpcMethods(NOTIFICATION_METHODS).find((candidate) => candidate.name === name) if (!found || 'stream' in found) { throw new Error(`${name} is not a one-shot RPC method`) } diff --git a/src/main/runtime/rpc/core-typed-method-contract.test.ts b/src/main/runtime/rpc/core-typed-method-contract.test.ts new file mode 100644 index 00000000000..fc1cdd1f97f --- /dev/null +++ b/src/main/runtime/rpc/core-typed-method-contract.test.ts @@ -0,0 +1,114 @@ +// The preserved types are the whole point of defineMethod, so they are asserted here: if a name +// widens to `string` or a result to `unknown`, these assertions fail at typecheck, not at runtime. +import { describe, expect, expectTypeOf, it } from 'vitest' +import { z } from 'zod' +import { + buildRegistry, + defineMethod, + defineStreamingMethod, + eraseRpcMethods, + isStreamingMethod, + type RpcContext, + type RpcMethod, + type RpcStreamingMethod +} from './core' +import type { ALL_RPC_METHODS } from './methods' +import { STATUS_METHODS } from './methods/status' +import type { HOST_CAPABILITY_METHODS } from './methods/host-capabilities' + +const ProbeParams = z.object({ id: z.string(), count: z.number().optional() }) + +const probe = defineMethod({ + name: 'test.typedProbe', + params: ProbeParams, + handler: (params) => ({ id: params.id, count: params.count ?? 0 }) +}) + +const schemalessProbe = defineMethod({ + name: 'test.schemalessProbe', + params: null, + handler: () => ['a', 'b'] +}) + +const streamingProbe = defineStreamingMethod({ + name: 'test.streamingProbe', + params: ProbeParams, + handler: async (params, _ctx, emit) => { + emit(params.id) + } +}) + +type ByName = Extract + +describe('defineMethod preserves the declared contract', () => { + it('keeps the literal method name', () => { + expectTypeOf(probe.name).toEqualTypeOf<'test.typedProbe'>() + expectTypeOf(streamingProbe.name).toEqualTypeOf<'test.streamingProbe'>() + expect(probe.name).toBe('test.typedProbe') + }) + + it('keeps the producer result type', () => { + expectTypeOf(probe.handler).returns.toEqualTypeOf<{ id: string; count: number }>() + expectTypeOf(schemalessProbe.handler).returns.toEqualTypeOf() + }) + + it('infers parsed params from the schema, and `void` without one', () => { + expectTypeOf(probe.handler) + .parameter(0) + .toEqualTypeOf<{ id: string; count?: number | undefined }>() + expectTypeOf(schemalessProbe.handler).parameter(0).toEqualTypeOf() + expectTypeOf(streamingProbe.handler) + .parameter(0) + .toEqualTypeOf<{ id: string; count?: number | undefined }>() + expectTypeOf(probe.params).toEqualTypeOf() + }) + + it('keeps a registered method addressable by its literal name', () => { + type StatusGet = ByName<(typeof STATUS_METHODS)[number], 'status.get'> + type ListDistros = ByName<(typeof HOST_CAPABILITY_METHODS)[number], 'host.wsl.listDistros'> + expectTypeOf().not.toBeNever() + expectTypeOf().returns.toExtend<{ runtimeId: string }>() + expectTypeOf().returns.toEqualTypeOf>() + // The manifest is the erasure boundary's input, so the literal names have to survive it too. + expectTypeOf>().not.toBeNever() + }) +}) + +describe('eraseRpcMethods is the registry boundary', () => { + it('erases to the shape the dispatcher calls, keeping the streaming split', () => { + expectTypeOf(eraseRpcMethods([probe])).toEqualTypeOf() + expectTypeOf(eraseRpcMethods([streamingProbe])).toEqualTypeOf() + expectTypeOf(eraseRpcMethods(STATUS_METHODS)).toEqualTypeOf() + expectTypeOf(eraseRpcMethods([probe])[0]!.handler) + .parameter(0) + .toEqualTypeOf() + }) + + it('returns the same methods, so nothing about the runtime value changes', () => { + const erased = eraseRpcMethods([probe, streamingProbe]) + + expect(erased[0]).toBe(probe) + expect(erased[1]).toBe(streamingProbe) + }) + + it('produces methods the registry accepts and the dispatcher can invoke', async () => { + const registry = buildRegistry([probe, streamingProbe, ...STATUS_METHODS]) + const registered = registry.get('test.typedProbe') + + expect(registered).toBe(probe) + expect(registry.get('status.get')).toBe(STATUS_METHODS[0]) + expect(isStreamingMethod(registry.get('test.streamingProbe')!)).toBe(true) + expect(registered && isStreamingMethod(registered)).toBe(false) + // The dispatcher parses params itself and then calls the erased handler with `unknown`. + const parsed: unknown = probe.params.parse({ id: 'a' }) + expect( + registered && !isStreamingMethod(registered) + ? await registered.handler(parsed, {} as RpcContext) + : undefined + ).toEqual({ id: 'a', count: 0 }) + }) + + it('rejects a duplicate name before erasure hides it', () => { + expect(() => buildRegistry([probe, probe])).toThrow('duplicate_rpc_method:test.typedProbe') + }) +}) diff --git a/src/main/runtime/rpc/core.ts b/src/main/runtime/rpc/core.ts index 702ea1b3aaa..c59c8da64d3 100644 --- a/src/main/runtime/rpc/core.ts +++ b/src/main/runtime/rpc/core.ts @@ -119,28 +119,27 @@ export type RpcContext = { ) => () => void } -export type RpcHandler = (params: TParams, ctx: RpcContext) => unknown +export type RpcHandler = (params: TParams, ctx: RpcContext) => TResult -// Why: RpcMethod erases the param type; centralizing the cast in defineMethod sidesteps RpcHandler's contravariance. -export type RpcMethod = { - readonly name: string - readonly params: ZodType | null - readonly handler: (params: unknown, ctx: RpcContext) => unknown +// Why: a schema-less method takes no params, so its handler must not be able to read the first argument. +type RpcParsedParams = TSchema extends ZodType + ? TSchema['_output'] + : void + +// Why: the authored shape — literal name, params schema, and producer result all survive for compile-time contracts. +export type RpcTypedMethod = { + readonly name: TName + readonly params: TSchema + readonly handler: RpcHandler, TResult> } -type DefineMethodSpec = { - name: string - params: TSchema - handler: RpcHandler -} - -export function defineMethod( - spec: DefineMethodSpec -): RpcMethod { +export function defineMethod( + spec: RpcTypedMethod +): RpcTypedMethod { return { name: spec.name, params: spec.params, - handler: spec.handler as RpcMethod['handler'] + handler: spec.handler } } @@ -150,6 +149,53 @@ export type RpcStreamingHandler = ( emit: (result: unknown) => void ) => Promise +// Why: emitted values stay `unknown` — the emit callback is an input, so there is no return position to infer them from. +export type RpcTypedStreamingMethod = { + readonly name: TName + readonly params: TSchema + readonly stream: true + readonly handler: RpcStreamingHandler> +} + +export function defineStreamingMethod( + spec: Omit, 'stream'> +): RpcTypedStreamingMethod { + return { + name: spec.name, + params: spec.params, + stream: true, + handler: spec.handler + } +} + +// Why `never` params: it makes the declaration a supertype of every parsed-params handler, so typed methods +// travel to the registry boundary — and only there get erased — without a cast in each methods module. +export type RpcMethodDeclaration = { + readonly name: string + readonly params: ZodType | null + readonly handler: (params: never, ctx: RpcContext) => unknown +} + +export type RpcStreamingMethodDeclaration = { + readonly name: string + readonly params: ZodType | null + readonly stream: true + readonly handler: ( + params: never, + ctx: RpcContext, + emit: (result: unknown) => void + ) => Promise +} + +export type RpcAnyMethodDeclaration = RpcMethodDeclaration | RpcStreamingMethodDeclaration + +// Why: RpcMethod is the registry's erased view; the dispatcher parses params itself and hands handlers `unknown`. +export type RpcMethod = { + readonly name: string + readonly params: ZodType | null + readonly handler: (params: unknown, ctx: RpcContext) => unknown +} + // Why: the `stream` flag lets the dispatcher route these to the emit-based path instead of the one-shot Promise path. export type RpcStreamingMethod = { readonly name: string @@ -162,34 +208,33 @@ export type RpcStreamingMethod = { ) => Promise } -type DefineStreamingMethodSpec = { - name: string - params: TSchema - handler: RpcStreamingHandler -} - -export function defineStreamingMethod( - spec: DefineStreamingMethodSpec -): RpcStreamingMethod { - return { - name: spec.name, - params: spec.params, - stream: true, - handler: spec.handler as RpcStreamingMethod['handler'] - } -} - export type RpcAnyMethod = RpcMethod | RpcStreamingMethod +// Why the overloads: erasure drops the parsed-params type, not the one-shot/streaming split the dispatcher routes on. +export function eraseRpcMethods(methods: readonly RpcMethodDeclaration[]): readonly RpcMethod[] +export function eraseRpcMethods( + methods: readonly RpcStreamingMethodDeclaration[] +): readonly RpcStreamingMethod[] +export function eraseRpcMethods( + methods: readonly RpcAnyMethodDeclaration[] +): readonly RpcAnyMethod[] +// Why: the one place the parsed-params type is dropped — contravariance makes it uncastable by assignment, and +// the dispatcher only ever calls a handler with an already-parsed `unknown`. Runtime value is untouched. +export function eraseRpcMethods( + methods: readonly RpcAnyMethodDeclaration[] +): readonly RpcAnyMethod[] { + return methods as readonly RpcAnyMethod[] +} + export function isStreamingMethod(method: RpcAnyMethod): method is RpcStreamingMethod { return 'stream' in method && method.stream === true } export type RpcRegistry = ReadonlyMap -export function buildRegistry(methods: readonly RpcAnyMethod[]): RpcRegistry { +export function buildRegistry(methods: readonly RpcAnyMethodDeclaration[]): RpcRegistry { const registry = new Map() - for (const method of methods) { + for (const method of eraseRpcMethods(methods)) { if (registry.has(method.name)) { throw new Error(`duplicate_rpc_method:${method.name}`) } diff --git a/src/main/runtime/rpc/dispatcher-request-parsing.ts b/src/main/runtime/rpc/dispatcher-request-parsing.ts index da4ec510e75..a4ada6df7c4 100644 --- a/src/main/runtime/rpc/dispatcher-request-parsing.ts +++ b/src/main/runtime/rpc/dispatcher-request-parsing.ts @@ -1,7 +1,7 @@ import { compile, type ZodType } from 'zod' import { formatZodError, - type RpcAnyMethod, + type RpcAnyMethodDeclaration, type RpcEnvelopeMeta, type RpcRequest, type RpcResponse @@ -12,7 +12,7 @@ const compiledParams = new WeakMap() export function parseRpcRequestParams( request: RpcRequest, - method: RpcAnyMethod, + method: RpcAnyMethodDeclaration, meta: RpcEnvelopeMeta ): { value: unknown; error?: undefined } | { value?: undefined; error: RpcResponse } { if (method.params === null) { diff --git a/src/main/runtime/rpc/dispatcher.ts b/src/main/runtime/rpc/dispatcher.ts index 73cfa596dd5..2c407207197 100644 --- a/src/main/runtime/rpc/dispatcher.ts +++ b/src/main/runtime/rpc/dispatcher.ts @@ -1,7 +1,7 @@ import { buildRegistry, isStreamingMethod, - type RpcAnyMethod, + type RpcAnyMethodDeclaration, type RpcEnvelopeMeta, type RpcRegistry, type RpcRequest, @@ -24,7 +24,10 @@ import { parseRpcRequestParams } from './dispatcher-request-parsing' import { RpcStreamingDispatcher } from './rpc-streaming-dispatcher' import { invokeDispatcherUnaryMethod } from './dispatcher-unary-method-invocation' -export type DispatcherOptions = { runtime: OrcaRuntimeService; methods?: readonly RpcAnyMethod[] } +export type DispatcherOptions = { + runtime: OrcaRuntimeService + methods?: readonly RpcAnyMethodDeclaration[] +} type DispatchCallOptions = RpcDispatchStreamingOptions diff --git a/src/main/runtime/rpc/methods/accounts.test.ts b/src/main/runtime/rpc/methods/accounts.test.ts index dc09f93fc22..ac8948422ac 100644 --- a/src/main/runtime/rpc/methods/accounts.test.ts +++ b/src/main/runtime/rpc/methods/accounts.test.ts @@ -2,11 +2,11 @@ import { describe, expect, it, vi } from 'vitest' import { tmpdir } from 'node:os' import { join } from 'node:path' import type { OrcaRuntimeService } from '../../orca-runtime' -import { isStreamingMethod } from '../core' +import { eraseRpcMethods, isStreamingMethod } from '../core' import { ACCOUNT_METHODS } from './accounts' function method(name: string) { - const found = ACCOUNT_METHODS.find((candidate) => candidate.name === name) + const found = eraseRpcMethods(ACCOUNT_METHODS).find((candidate) => candidate.name === name) if (!found) { throw new Error(`Missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/accounts.ts b/src/main/runtime/rpc/methods/accounts.ts index 328276518d7..f7fe0af90ec 100644 --- a/src/main/runtime/rpc/methods/accounts.ts +++ b/src/main/runtime/rpc/methods/accounts.ts @@ -1,4 +1,4 @@ -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { AccountsUnsubscribeParams, AddClaudeFromConfigDirParams, @@ -24,7 +24,7 @@ let accountsSubscriptionSeq = 0 // captures an already-authenticated CLAUDE_CONFIG_DIR (no PTY) so the local // `orca account add` CLI can register accounts on a headless host; it is gated // to the local runtime connection, never a mobile device token. See #1438. -export const ACCOUNT_METHODS: readonly RpcAnyMethod[] = [ +export const ACCOUNT_METHODS = [ defineMethod({ name: 'accounts.list', params: ListAccountsParams, diff --git a/src/main/runtime/rpc/methods/agent-hooks.test.ts b/src/main/runtime/rpc/methods/agent-hooks.test.ts index 5781045a1f7..f78e7709d6e 100644 --- a/src/main/runtime/rpc/methods/agent-hooks.test.ts +++ b/src/main/runtime/rpc/methods/agent-hooks.test.ts @@ -1,6 +1,6 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import type { OrcaRuntimeService } from '../../orca-runtime' -import { isStreamingMethod, type RpcContext } from '../core' +import { eraseRpcMethods, isStreamingMethod, type RpcContext } from '../core' const { installForRuntimeHomeSerializedMock, realpathMock } = vi.hoisted(() => ({ installForRuntimeHomeSerializedMock: vi.fn(), @@ -23,7 +23,7 @@ const RUNTIME_HOME = '\\\\wsl.localhost\\Ubuntu-24.04\\home\\jin\\.local\\share\\orca\\codex-runtime-home\\home' function prepareMethod() { - const method = AGENT_HOOK_METHODS.find( + const method = eraseRpcMethods(AGENT_HOOK_METHODS).find( (candidate) => candidate.name === 'agentHooks.prepareCodexForWslPane' ) if (!method || isStreamingMethod(method)) { diff --git a/src/main/runtime/rpc/methods/agent-hooks.ts b/src/main/runtime/rpc/methods/agent-hooks.ts index e68c7df05df..4d9f4a7706c 100644 --- a/src/main/runtime/rpc/methods/agent-hooks.ts +++ b/src/main/runtime/rpc/methods/agent-hooks.ts @@ -1,8 +1,8 @@ import { prepareManagedWslCodexHomeBeforeShellLaunch } from '../../../codex/managed-wsl-home-shell-preflight' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { PrepareCodexForWslPaneParams } from '../../../../shared/rpc-contract/agent-hooks-params' -export const AGENT_HOOK_METHODS: readonly RpcMethod[] = [ +export const AGENT_HOOK_METHODS = [ defineMethod({ name: 'agentHooks.prepareCodexForWslPane', params: PrepareCodexForWslPaneParams, diff --git a/src/main/runtime/rpc/methods/agent-session.ts b/src/main/runtime/rpc/methods/agent-session.ts index 79da783a939..84e008a7f55 100644 --- a/src/main/runtime/rpc/methods/agent-session.ts +++ b/src/main/runtime/rpc/methods/agent-session.ts @@ -10,7 +10,7 @@ import { parseAgentSessionOperationTimestamp } from '../../../../shared/agent-session-host-authority' import type { OrcaRuntimeService } from '../../orca-runtime' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { CreateAgentSessionParams, EnsureAgentSessionParams @@ -58,7 +58,7 @@ function assertOperationTimestampWithinFutureSkew(clientOperationId: string): vo } } -export const AGENT_SESSION_METHODS: RpcAnyMethod[] = [ +export const AGENT_SESSION_METHODS = [ defineMethod({ name: 'terminal.ensureAgentSession', params: EnsureAgentSessionParams, diff --git a/src/main/runtime/rpc/methods/ai-vault.ts b/src/main/runtime/rpc/methods/ai-vault.ts index d5f52bafded..abc3aff2650 100644 --- a/src/main/runtime/rpc/methods/ai-vault.ts +++ b/src/main/runtime/rpc/methods/ai-vault.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { restampAiVaultListResult } from '../../../ai-vault/session-list-results' import type { AiVaultPrepareSessionResumeArgs } from '../../../../shared/ai-vault-resume-preparation' import { LOCAL_EXECUTION_HOST_ID } from '../../../../shared/execution-host' @@ -15,7 +15,7 @@ import { } from '../../../../shared/rpc-contract/ai-vault-params' export { AiVaultListSessionsParams, AiVaultPrepareSessionResumeParams, AiVaultSessionTitlesParams } -export const AI_VAULT_METHODS: RpcMethod[] = [ +export const AI_VAULT_METHODS = [ defineMethod({ name: 'aiVault.resolveSessionTitles', params: AiVaultSessionTitlesParams, diff --git a/src/main/runtime/rpc/methods/artifacts.ts b/src/main/runtime/rpc/methods/artifacts.ts index c627380d612..b6b91af8b0b 100644 --- a/src/main/runtime/rpc/methods/artifacts.ts +++ b/src/main/runtime/rpc/methods/artifacts.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ArtifactsDeleteParams, ListOptions, @@ -6,7 +6,7 @@ import { WriteRequest } from '../../../../shared/rpc-contract/artifacts-params' -export const ARTIFACT_METHODS: readonly RpcAnyMethod[] = [ +export const ARTIFACT_METHODS = [ defineMethod({ name: 'artifacts.list', params: ListOptions, diff --git a/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts b/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts index f1d2081dc6b..6af7ba467e9 100644 --- a/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts +++ b/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts @@ -4,14 +4,14 @@ * current callers also receive owner metadata. */ import { describe, expect, it, vi } from 'vitest' -import type { RpcContext, RpcRequest } from '../core' +import { eraseRpcMethods, type RpcContext, type RpcRequest } from '../core' import { RpcDispatcher } from '../dispatcher' import type { OrcaRuntimeService } from '../../orca-runtime' import { AUTOMATION_METHODS } from './automations' import { AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' function method(name: string) { - const found = AUTOMATION_METHODS.find((entry) => entry.name === name) + const found = eraseRpcMethods(AUTOMATION_METHODS).find((entry) => entry.name === name) if (!found?.params) { throw new Error(`missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/automations.ts b/src/main/runtime/rpc/methods/automations.ts index ff5daca315c..b1e3eebe6eb 100644 --- a/src/main/runtime/rpc/methods/automations.ts +++ b/src/main/runtime/rpc/methods/automations.ts @@ -1,6 +1,6 @@ import { AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { AutomationOwnerPrecondition } from '../../../../shared/automation-owner-precondition' -import { defineMethod, type RpcContext, type RpcMethod } from '../core' +import { defineMethod, type RpcContext } from '../core' import { AutomationCreate, AutomationId, @@ -25,7 +25,7 @@ function mutationOwner( return context.runtime.automationOwnerPrecondition(id) ?? undefined } -export const AUTOMATION_METHODS: RpcMethod[] = [ +export const AUTOMATION_METHODS = [ defineMethod({ name: 'automation.list', params: AutomationList, diff --git a/src/main/runtime/rpc/methods/browser-client-file-channel.ts b/src/main/runtime/rpc/methods/browser-client-file-channel.ts index b2020488af2..362a1990531 100644 --- a/src/main/runtime/rpc/methods/browser-client-file-channel.ts +++ b/src/main/runtime/rpc/methods/browser-client-file-channel.ts @@ -7,7 +7,7 @@ import { BROWSER_CLIENT_HOST_RUNTIME_CAPABILITY } from '../../../../shared/proto import { getBrowserClientDownloadTransferStore } from '../../browser-client-download-transfer-store' import { getBrowserHostLeaseRegistry } from '../../browser-host-lease-registry-instance' import { getRuntimeBrowserPageRegistry } from '../../runtime-browser-page-registry' -import { defineMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, type RpcContext } from '../core' type FileChannelAuthorityParams = { browserHostClientId: string @@ -64,7 +64,7 @@ function requireFileChannelPage( return page } -export const BROWSER_CLIENT_FILE_CHANNEL_METHODS: RpcAnyMethod[] = [ +export const BROWSER_CLIENT_FILE_CHANNEL_METHODS = [ defineMethod({ name: 'browser.clientHost.fileChannel.read', params: BrowserClientFileChannelReadParams, diff --git a/src/main/runtime/rpc/methods/browser-client-host.ts b/src/main/runtime/rpc/methods/browser-client-host.ts index 525fd4fde96..e06632f7959 100644 --- a/src/main/runtime/rpc/methods/browser-client-host.ts +++ b/src/main/runtime/rpc/methods/browser-client-host.ts @@ -14,9 +14,9 @@ import { getRuntimeBrowserPageRegistry } from '../../runtime-browser-page-regist import { adoptRuntimeBrowserClientPagesFromInventory } from '../../runtime-browser-client-page-adoption' import { recoverUnavailableRuntimeBrowserClientPages } from '../../runtime-browser-client-page-recovery' import { releaseRuntimeBrowserClientPageRecord } from '../../runtime-browser-client-page-release' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' -export const BROWSER_CLIENT_HOST_METHODS: RpcAnyMethod[] = [ +export const BROWSER_CLIENT_HOST_METHODS = [ defineStreamingMethod({ name: 'browser.clientHost.attach', params: BrowserClientHostAttachParams, diff --git a/src/main/runtime/rpc/methods/browser-core.ts b/src/main/runtime/rpc/methods/browser-core.ts index 596c30a31c8..780f40fb373 100644 --- a/src/main/runtime/rpc/methods/browser-core.ts +++ b/src/main/runtime/rpc/methods/browser-core.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { BrowserTarget } from '../schemas' import { Check, @@ -35,7 +35,7 @@ import { BrowserOpenUrlParams, BrowserTabCreateParams } from './browser-tab-crea import { BROWSER_TEXT_METHODS } from './browser-text-rpc-methods' import { CertificateProceed } from '../../../../shared/rpc-contract/browser-core-params' -export const BROWSER_CORE_METHODS: RpcMethod[] = [ +export const BROWSER_CORE_METHODS = [ defineMethod({ name: 'browser.snapshot', params: BrowserTarget, diff --git a/src/main/runtime/rpc/methods/browser-extras.ts b/src/main/runtime/rpc/methods/browser-extras.ts index fe87bde950f..000838c4938 100644 --- a/src/main/runtime/rpc/methods/browser-extras.ts +++ b/src/main/runtime/rpc/methods/browser-extras.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { assertRpcClipboardTextWriteWithinLimit } from '../rpc-clipboard-text-validation' import { BrowserTarget } from '../schemas' import { @@ -23,7 +23,7 @@ import { } from './browser-schemas' import { MouseClick } from '../../../../shared/rpc-contract/browser-extras-params' -export const BROWSER_EXTRA_METHODS: RpcMethod[] = [ +export const BROWSER_EXTRA_METHODS = [ defineMethod({ name: 'browser.cookie.get', params: CookieGet, diff --git a/src/main/runtime/rpc/methods/browser-network-tunnel.ts b/src/main/runtime/rpc/methods/browser-network-tunnel.ts index 405419deef0..d610824b1f2 100644 --- a/src/main/runtime/rpc/methods/browser-network-tunnel.ts +++ b/src/main/runtime/rpc/methods/browser-network-tunnel.ts @@ -12,14 +12,14 @@ import { BROWSER_NETWORK_TUNNEL_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { getBrowserHostLeaseRegistry } from '../../browser-host-lease-registry-instance' -import { defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineStreamingMethod } from '../core' const outboundMemoryBudgets = new BrowserNetworkTunnelOutboundMemoryBudgetRegistry() export function createBrowserNetworkTunnelMethods( memoryBudgets: BrowserNetworkTunnelOutboundMemoryBudgetRegistry = outboundMemoryBudgets, resolveExecutionRoute: BrowserNetworkExecutionRouteResolver = resolveBrowserNetworkExecutionRoute -): RpcAnyMethod[] { +) { return [ defineStreamingMethod({ name: 'network.browserTunnel', diff --git a/src/main/runtime/rpc/methods/browser-screencast.ts b/src/main/runtime/rpc/methods/browser-screencast.ts index ed16e9d0edf..798e1ed84d2 100644 --- a/src/main/runtime/rpc/methods/browser-screencast.ts +++ b/src/main/runtime/rpc/methods/browser-screencast.ts @@ -1,11 +1,11 @@ -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { Screencast } from './browser-schemas' import { BrowserError } from '../../../browser/browser-error' import { BROWSER_UNAVAILABLE_ERROR_CODE } from '../../../../shared/runtime-types' import { runtimeBrowserCommandsFactoryIsAvailable } from '../../runtime-browser-commands-factory' import { ScreencastUnsubscribe } from '../../../../shared/rpc-contract/browser-screencast-params' -export const BROWSER_SCREENCAST_METHODS: RpcAnyMethod[] = [ +export const BROWSER_SCREENCAST_METHODS = [ defineStreamingMethod({ name: 'browser.screencast', params: Screencast, diff --git a/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts b/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts index 813dbe9d2e2..18fd19a0e6a 100644 --- a/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts +++ b/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts @@ -1,8 +1,8 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { assertRpcClipboardTextWriteWithinLimit } from '../rpc-clipboard-text-validation' import { Fill, KeyboardInsert, Type } from './browser-schemas' -export const BROWSER_TEXT_METHODS: RpcMethod[] = [ +export const BROWSER_TEXT_METHODS = [ defineMethod({ name: 'browser.fill', params: Fill, diff --git a/src/main/runtime/rpc/methods/client-events.test.ts b/src/main/runtime/rpc/methods/client-events.test.ts index bf5026c819e..cae682ea4b4 100644 --- a/src/main/runtime/rpc/methods/client-events.test.ts +++ b/src/main/runtime/rpc/methods/client-events.test.ts @@ -1,11 +1,16 @@ import { describe, expect, it, vi } from 'vitest' import type { RuntimeClientEvent } from '../../../../shared/runtime-client-events' import type { OrcaRuntimeService } from '../../orca-runtime' -import { isStreamingMethod, type RpcContext, type RpcStreamingMethod } from '../core' +import { + eraseRpcMethods, + isStreamingMethod, + type RpcContext, + type RpcStreamingMethod +} from '../core' // Why: importing client-events directly trips its module-init cycle through ipc/ssh; the index resolves it. import { ALL_RPC_METHODS } from './index' -const subscribeMethod = ALL_RPC_METHODS.find( +const subscribeMethod = eraseRpcMethods(ALL_RPC_METHODS).find( (method) => method.name === 'runtime.clientEvents.subscribe' && isStreamingMethod(method) ) as RpcStreamingMethod diff --git a/src/main/runtime/rpc/methods/client-events.ts b/src/main/runtime/rpc/methods/client-events.ts index 6c23ed9f15d..e7506ad2f59 100644 --- a/src/main/runtime/rpc/methods/client-events.ts +++ b/src/main/runtime/rpc/methods/client-events.ts @@ -1,11 +1,11 @@ import { getRegisteredSshState, listRegisteredSshTargets } from '../../../ssh/ssh-target-registry' import { getPublicSshState } from '../../public-ssh-state' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { ClientEventsUnsubscribeParams } from '../../../../shared/rpc-contract/client-events-params' let clientEventSubscriptionSeq = 0 -export const CLIENT_EVENT_METHODS: readonly RpcAnyMethod[] = [ +export const CLIENT_EVENT_METHODS = [ defineStreamingMethod({ name: 'runtime.clientEvents.subscribe', params: null, diff --git a/src/main/runtime/rpc/methods/client-ui.ts b/src/main/runtime/rpc/methods/client-ui.ts index ffd964b6be6..6ed36a6fe83 100644 --- a/src/main/runtime/rpc/methods/client-ui.ts +++ b/src/main/runtime/rpc/methods/client-ui.ts @@ -1,6 +1,6 @@ import { omitPairingLocalUiFields } from '../../../../shared/pairing-local-ui-fields' import type { PersistedUIState } from '../../../../shared/persisted-ui-state-types' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { NativeChatSessionOptionsMutation, PRBotAuthorOverrideUpdate, @@ -12,7 +12,7 @@ import { FeatureInteractionIdParam, UiUpdate } from './client-ui-schemas' import { TerminalQuickCommandsUpdate } from './terminal-quick-command-rpc-schema' -export const CLIENT_UI_METHODS: RpcMethod[] = [ +export const CLIENT_UI_METHODS = [ defineMethod({ name: 'settings.get', params: null, diff --git a/src/main/runtime/rpc/methods/clipboard.ts b/src/main/runtime/rpc/methods/clipboard.ts index e27b1cf1607..3ec78265dc6 100644 --- a/src/main/runtime/rpc/methods/clipboard.ts +++ b/src/main/runtime/rpc/methods/clipboard.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcContext, type RpcMethod } from '../core' +import { defineMethod, type RpcContext } from '../core' import { saveClipboardImageBufferAsTempFile } from '../../../window/clipboard-image-temp-file' import { randomUUID } from 'node:crypto' import { recordMobileClipboardImagePath } from '../mobile-clipboard-image-provenance' @@ -95,7 +95,7 @@ function assertValidBase64Content(value: string): void { } } -export const CLIPBOARD_METHODS: RpcMethod[] = [ +export const CLIPBOARD_METHODS = [ defineMethod({ name: 'clipboard.saveImageAsTempFile', params: SaveImageAsTempFile, diff --git a/src/main/runtime/rpc/methods/computer-actions.test.ts b/src/main/runtime/rpc/methods/computer-actions.test.ts index 96dc374e459..fa2b7d83ee1 100644 --- a/src/main/runtime/rpc/methods/computer-actions.test.ts +++ b/src/main/runtime/rpc/methods/computer-actions.test.ts @@ -27,6 +27,7 @@ vi.mock('../../../computer/macos-computer-use-permissions', () => ({ })) import { COMPUTER_METHODS, resetComputerSessionsForTest } from './computer' +import { eraseRpcMethods } from '../core' describe('computer action RPC methods', () => { beforeEach(() => { @@ -269,7 +270,7 @@ describe('computer action RPC methods', () => { }) function findMethod(name: string) { - const method = COMPUTER_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(COMPUTER_METHODS).find((candidate) => candidate.name === name) if (!method) { throw new Error(`missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/computer.test.ts b/src/main/runtime/rpc/methods/computer.test.ts index 6a01fac7d0a..073a1c0a363 100644 --- a/src/main/runtime/rpc/methods/computer.test.ts +++ b/src/main/runtime/rpc/methods/computer.test.ts @@ -1,5 +1,5 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' -import { buildRegistry } from '../core' +import { eraseRpcMethods, buildRegistry } from '../core' import { CLIPBOARD_TEXT_WRITE_MAX_BYTES } from '../../../../shared/clipboard-text' const computerMocks = vi.hoisted(() => ({ @@ -249,7 +249,7 @@ describe('computer RPC methods', () => { }) function findMethod(name: string) { - const method = COMPUTER_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(COMPUTER_METHODS).find((candidate) => candidate.name === name) if (!method) { throw new Error(`missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/computer.ts b/src/main/runtime/rpc/methods/computer.ts index 07459427bd4..e2708667633 100644 --- a/src/main/runtime/rpc/methods/computer.ts +++ b/src/main/runtime/rpc/methods/computer.ts @@ -6,7 +6,7 @@ import { callComputerSidecarSnapshot, resetComputerSidecarForTest } from '../../../computer/sidecar-client' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { Click, ComputerObserveTarget, @@ -31,7 +31,7 @@ export function resetComputerSessionsForTest(): void { resetComputerSidecarForTest() } -export const COMPUTER_METHODS: RpcMethod[] = [ +export const COMPUTER_METHODS = [ defineMethod({ name: 'computer.capabilities', params: ComputerCapabilitiesParams, diff --git a/src/main/runtime/rpc/methods/diagnostics.ts b/src/main/runtime/rpc/methods/diagnostics.ts index 4d158d98f63..953eccd5d51 100644 --- a/src/main/runtime/rpc/methods/diagnostics.ts +++ b/src/main/runtime/rpc/methods/diagnostics.ts @@ -1,6 +1,6 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' -export const DIAGNOSTICS_METHODS: RpcMethod[] = [ +export const DIAGNOSTICS_METHODS = [ defineMethod({ name: 'diagnostics.memory', params: null, diff --git a/src/main/runtime/rpc/methods/emulator.ts b/src/main/runtime/rpc/methods/emulator.ts index 00c239658d9..b472e539603 100644 --- a/src/main/runtime/rpc/methods/emulator.ts +++ b/src/main/runtime/rpc/methods/emulator.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import path from 'node:path' import { z } from 'zod' import { @@ -32,7 +32,7 @@ const InstallParams = z.object({ worktree: z.string().optional() }) -export const EMULATOR_METHODS: RpcMethod[] = [ +export const EMULATOR_METHODS = [ defineMethod({ name: 'emulator.list', params: ListParams, diff --git a/src/main/runtime/rpc/methods/files-mutation-methods.ts b/src/main/runtime/rpc/methods/files-mutation-methods.ts index 3c7e4231b96..eb734d5c8d6 100644 --- a/src/main/runtime/rpc/methods/files-mutation-methods.ts +++ b/src/main/runtime/rpc/methods/files-mutation-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { FileCommitUpload, FileCopy, @@ -33,7 +33,7 @@ function sshMutationArguments( ] } -export const FILE_MUTATION_METHODS: RpcAnyMethod[] = [ +export const FILE_MUTATION_METHODS = [ defineMethod({ name: 'files.write', params: FileWrite, diff --git a/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts b/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts index cc343dd3ca1..20d52eb424d 100644 --- a/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts +++ b/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts @@ -1,11 +1,11 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { remoteFileContentBudget } from './files-remote-content-budget' import { TerminalArtifactFile, TerminalArtifactFileWrite } from '../../../../shared/rpc-contract/files-terminal-artifact-params' -export const FILE_TERMINAL_ARTIFACT_METHODS: RpcAnyMethod[] = [ +export const FILE_TERMINAL_ARTIFACT_METHODS = [ defineMethod({ name: 'files.readTerminalArtifact', params: TerminalArtifactFile, diff --git a/src/main/runtime/rpc/methods/files.ts b/src/main/runtime/rpc/methods/files.ts index 6f5cdbe4b34..055c02b3265 100644 --- a/src/main/runtime/rpc/methods/files.ts +++ b/src/main/runtime/rpc/methods/files.ts @@ -1,4 +1,4 @@ -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { runFileWatchStream } from './file-watch-stream-lifecycle' import { FILE_MUTATION_METHODS } from './files-mutation-methods' import { remoteFileContentBudget } from './files-remote-content-budget' @@ -21,7 +21,7 @@ import { let filesWatchSubscriptionSeq = 0 -export const FILE_METHODS: RpcAnyMethod[] = [ +export const FILE_METHODS = [ defineMethod({ name: 'files.list', params: WorktreeSelector, diff --git a/src/main/runtime/rpc/methods/folder-workspace.ts b/src/main/runtime/rpc/methods/folder-workspace.ts index e0f780e710a..a2e178b8ed9 100644 --- a/src/main/runtime/rpc/methods/folder-workspace.ts +++ b/src/main/runtime/rpc/methods/folder-workspace.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { resolveRpcWorkspaceCreatorProvenance } from '../workspace-creator-context' import { FolderWorkspaceCreate, @@ -7,7 +7,7 @@ import { FolderWorkspaceUpdate } from '../../../../shared/rpc-contract/folder-workspace-params' -export const FOLDER_WORKSPACE_METHODS: RpcMethod[] = [ +export const FOLDER_WORKSPACE_METHODS = [ defineMethod({ name: 'folderWorkspace.list', params: null, diff --git a/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts b/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts index 787c8581dea..1dffcfb2211 100644 --- a/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts +++ b/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import type { GlobalSettings } from '../../../../shared/global-settings-types' import type { ResolvedSourceControlAiGenerationParams } from '../../../../shared/source-control-ai' import { @@ -58,7 +58,7 @@ function buildCommitMessageGenerationOverride(params: { } } -export const GIT_COMMIT_MESSAGE_GENERATION_METHODS: RpcMethod[] = [ +export const GIT_COMMIT_MESSAGE_GENERATION_METHODS = [ defineMethod({ name: 'git.generateCommitMessage', params: GitGenerateCommitMessage, diff --git a/src/main/runtime/rpc/methods/git-diff-methods.ts b/src/main/runtime/rpc/methods/git-diff-methods.ts index 7b5c0661c0e..edcaebbdf42 100644 --- a/src/main/runtime/rpc/methods/git-diff-methods.ts +++ b/src/main/runtime/rpc/methods/git-diff-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { remoteRpcContentBudget } from '../../../../shared/remote-rpc-content-budget' import { GitBranchDiff, GitCommitDiff, GitDiff } from './git-params' @@ -11,7 +11,7 @@ function remoteDiffContentBudget( return clientKind && requestId ? remoteRpcContentBudget(requestId) : undefined } -export const GIT_DIFF_METHODS: RpcMethod[] = [ +export const GIT_DIFF_METHODS = [ defineMethod({ name: 'git.diff', params: GitDiff, diff --git a/src/main/runtime/rpc/methods/git.ts b/src/main/runtime/rpc/methods/git.ts index ddfbe7bf273..20102d71e28 100644 --- a/src/main/runtime/rpc/methods/git.ts +++ b/src/main/runtime/rpc/methods/git.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { GIT_COMMIT_MESSAGE_GENERATION_METHODS } from './git-commit-message-generation-methods' import { GIT_DIFF_METHODS } from './git-diff-methods' import { @@ -21,7 +21,7 @@ import { WorktreeSelector } from './git-params' -export const GIT_METHODS: RpcMethod[] = [ +export const GIT_METHODS = [ defineMethod({ name: 'git.status', params: GitStatusParams, diff --git a/src/main/runtime/rpc/methods/github-issue-methods.ts b/src/main/runtime/rpc/methods/github-issue-methods.ts index 40ed162f28b..86300017741 100644 --- a/src/main/runtime/rpc/methods/github-issue-methods.ts +++ b/src/main/runtime/rpc/methods/github-issue-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { CreateIssue, Issue, @@ -6,7 +6,7 @@ import { UpdateIssue } from '../../../../shared/rpc-contract/github-issue-params' -export const GITHUB_ISSUE_METHODS: RpcMethod[] = [ +export const GITHUB_ISSUE_METHODS = [ defineMethod({ name: 'github.issue', params: Issue, diff --git a/src/main/runtime/rpc/methods/github-project-methods.ts b/src/main/runtime/rpc/methods/github-project-methods.ts index 4cd2641b11a..c05086decbf 100644 --- a/src/main/runtime/rpc/methods/github-project-methods.ts +++ b/src/main/runtime/rpc/methods/github-project-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { SlugRepo } from './github-repo-target-schemas' import { ClearProjectItemField, @@ -17,7 +17,7 @@ import { SlugPullRequestUpdate } from '../../../../shared/rpc-contract/github-project-params' -export const GITHUB_PROJECT_METHODS: RpcMethod[] = [ +export const GITHUB_PROJECT_METHODS = [ defineMethod({ name: 'github.project.listAccessible', params: GithubProjectListAccessibleParams, diff --git a/src/main/runtime/rpc/methods/github-pull-request-methods.ts b/src/main/runtime/rpc/methods/github-pull-request-methods.ts index f7e66d41008..0a7f2efd522 100644 --- a/src/main/runtime/rpc/methods/github-pull-request-methods.ts +++ b/src/main/runtime/rpc/methods/github-pull-request-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { PRCommentReaction, PrForBranch, @@ -11,7 +11,7 @@ import { ReviewThread } from '../../../../shared/rpc-contract/github-pull-request-params' -export const GITHUB_PULL_REQUEST_METHODS: RpcMethod[] = [ +export const GITHUB_PULL_REQUEST_METHODS = [ defineMethod({ name: 'github.prForBranch', params: PrForBranch, diff --git a/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts b/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts index ca5bdcabbab..59089b82015 100644 --- a/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts +++ b/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { MarkPrReadyForReview, MergePr, @@ -12,7 +12,7 @@ import { UpdatePrTitle } from '../../../../shared/rpc-contract/github-pull-request-update-params' -export const GITHUB_PULL_REQUEST_UPDATE_METHODS: RpcMethod[] = [ +export const GITHUB_PULL_REQUEST_UPDATE_METHODS = [ defineMethod({ name: 'github.updatePRTitle', params: UpdatePrTitle, diff --git a/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts b/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts index d4722ec1466..6b2d83988d7 100644 --- a/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts +++ b/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { RepoSelector } from './github-repo-target-schemas' import { IssuesList, @@ -10,7 +10,7 @@ import { WorkItemsList } from '../../../../shared/rpc-contract/github-repo-work-item-params' -export const GITHUB_REPO_WORK_ITEM_METHODS: RpcMethod[] = [ +export const GITHUB_REPO_WORK_ITEM_METHODS = [ defineMethod({ name: 'github.repoSlug', params: RepoSelector, diff --git a/src/main/runtime/rpc/methods/github.ts b/src/main/runtime/rpc/methods/github.ts index 1dd4cb28823..d2113f9ac22 100644 --- a/src/main/runtime/rpc/methods/github.ts +++ b/src/main/runtime/rpc/methods/github.ts @@ -1,11 +1,10 @@ -import type { RpcMethod } from '../core' import { GITHUB_ISSUE_METHODS } from './github-issue-methods' import { GITHUB_PROJECT_METHODS } from './github-project-methods' import { GITHUB_PULL_REQUEST_METHODS } from './github-pull-request-methods' import { GITHUB_PULL_REQUEST_UPDATE_METHODS } from './github-pull-request-update-methods' import { GITHUB_REPO_WORK_ITEM_METHODS } from './github-repo-work-item-methods' -export const GITHUB_METHODS: RpcMethod[] = [ +export const GITHUB_METHODS = [ ...GITHUB_REPO_WORK_ITEM_METHODS, ...GITHUB_ISSUE_METHODS, ...GITHUB_PULL_REQUEST_METHODS, diff --git a/src/main/runtime/rpc/methods/gitlab.ts b/src/main/runtime/rpc/methods/gitlab.ts index ba840447cfd..93f73d5f05c 100644 --- a/src/main/runtime/rpc/methods/gitlab.ts +++ b/src/main/runtime/rpc/methods/gitlab.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { normalizeGitLabIssueListArgs } from '../../../gitlab/gitlab-preload-args' import { toGitLabJobLogExcerptResult } from '../../../../shared/gitlab-job-log-excerpt' import { @@ -23,7 +23,7 @@ import { WorkItemsList } from '../../../../shared/rpc-contract/gitlab-params' -export const GITLAB_METHODS: RpcMethod[] = [ +export const GITLAB_METHODS = [ defineMethod({ name: 'gitlab.listMRs', params: WorkItemsList, diff --git a/src/main/runtime/rpc/methods/host-capabilities.ts b/src/main/runtime/rpc/methods/host-capabilities.ts index afa1af88474..85a32fd1e5c 100644 --- a/src/main/runtime/rpc/methods/host-capabilities.ts +++ b/src/main/runtime/rpc/methods/host-capabilities.ts @@ -1,9 +1,9 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { isPwshAvailableAsync } from '../../../pwsh' import { isWslAvailableAsync, listWslDistrosAsync } from '../../../wsl' import { isGitBashAvailable } from '../../../git-bash' -export const HOST_CAPABILITY_METHODS: RpcMethod[] = [ +export const HOST_CAPABILITY_METHODS = [ defineMethod({ name: 'host.platform', params: null, diff --git a/src/main/runtime/rpc/methods/hosted-review.ts b/src/main/runtime/rpc/methods/hosted-review.ts index 84b297ffa4e..51d663210d8 100644 --- a/src/main/runtime/rpc/methods/hosted-review.ts +++ b/src/main/runtime/rpc/methods/hosted-review.ts @@ -1,11 +1,11 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { HostedReviewCreate, HostedReviewCreationEligibility, HostedReviewForBranch } from '../../../../shared/rpc-contract/hosted-review-params' -export const HOSTED_REVIEW_METHODS: RpcMethod[] = [ +export const HOSTED_REVIEW_METHODS = [ defineMethod({ name: 'hostedReview.forBranch', params: HostedReviewForBranch, diff --git a/src/main/runtime/rpc/methods/index.ts b/src/main/runtime/rpc/methods/index.ts index ba77b94803e..3a53bccb9ce 100644 --- a/src/main/runtime/rpc/methods/index.ts +++ b/src/main/runtime/rpc/methods/index.ts @@ -1,4 +1,3 @@ -import type { RpcAnyMethod } from '../core' import { STATUS_METHODS } from './status' import { AI_VAULT_METHODS } from './ai-vault' import { AUTOMATION_METHODS } from './automations' @@ -50,7 +49,7 @@ import { AGENT_HOOK_METHODS } from './agent-hooks' // Why: a flat manifest keeps registration order explicit and provides one // grep-point for "what methods does the RPC server expose?" — useful when // auditing the security boundary or wiring new CLI commands. -export const ALL_RPC_METHODS: readonly RpcAnyMethod[] = [ +export const ALL_RPC_METHODS = [ ...STATUS_METHODS, ...AGENT_HOOK_METHODS, ...AI_VAULT_METHODS, diff --git a/src/main/runtime/rpc/methods/jira.ts b/src/main/runtime/rpc/methods/jira.ts index 4463c969919..0e959cf0f5e 100644 --- a/src/main/runtime/rpc/methods/jira.ts +++ b/src/main/runtime/rpc/methods/jira.ts @@ -2,7 +2,7 @@ import { JIRA_PAYLOAD_CHUNK_CHARS, JIRA_PAYLOAD_MAX_CHARS } from '../../../../shared/jira-payload-stream' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { AssignableUsers, Connect, @@ -34,7 +34,7 @@ function emitJiraPayload(value: unknown, emit: (result: unknown) => void): void emit({ type: 'end' }) } -export const JIRA_METHODS: RpcAnyMethod[] = [ +export const JIRA_METHODS = [ defineMethod({ name: 'jira.connect', params: Connect, diff --git a/src/main/runtime/rpc/methods/linear-agent-access.ts b/src/main/runtime/rpc/methods/linear-agent-access.ts index e76c843dede..2c46face39d 100644 --- a/src/main/runtime/rpc/methods/linear-agent-access.ts +++ b/src/main/runtime/rpc/methods/linear-agent-access.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { linearError } from '../../../linear/issue-context-errors' import { isLinearUuid } from '../../../../shared/linear/uuid' import { @@ -28,7 +28,7 @@ function parseLinearWriteId(writeId: string | undefined): string | undefined { return writeId } -export const LINEAR_AGENT_ACCESS_METHODS: RpcMethod[] = [ +export const LINEAR_AGENT_ACCESS_METHODS = [ defineMethod({ name: 'linear.saveIssue', params: LinearSaveIssue, diff --git a/src/main/runtime/rpc/methods/linear-project-create.ts b/src/main/runtime/rpc/methods/linear-project-create.ts index f2c020f6656..8377b69acdf 100644 --- a/src/main/runtime/rpc/methods/linear-project-create.ts +++ b/src/main/runtime/rpc/methods/linear-project-create.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { CreateProject } from '../../../../shared/rpc-contract/linear-project-create-params' -export const LINEAR_PROJECT_CREATE_METHOD: RpcMethod = defineMethod({ +export const LINEAR_PROJECT_CREATE_METHOD = defineMethod({ name: 'linear.createProject', params: CreateProject, handler: async (params, { runtime }) => diff --git a/src/main/runtime/rpc/methods/linear.ts b/src/main/runtime/rpc/methods/linear.ts index f0ec2106830..456b4c5b4e9 100644 --- a/src/main/runtime/rpc/methods/linear.ts +++ b/src/main/runtime/rpc/methods/linear.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { LINEAR_PROJECT_CREATE_METHOD } from './linear-project-create' import { LINEAR_ISSUE_LIST_METHOD, LINEAR_MCP_ISSUE_LIST_METHOD } from './linear-issue-list-method' import { @@ -20,7 +20,7 @@ import { WorkspaceSelection } from '../../../../shared/rpc-contract/linear-params' -export const LINEAR_METHODS: RpcMethod[] = [ +export const LINEAR_METHODS = [ defineMethod({ name: 'linear.connect', params: Connect, diff --git a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts index dcba8b7b64e..756d4719f22 100644 --- a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts +++ b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' -export const MOBILE_MARKDOWN_TAB_METHODS: RpcAnyMethod[] = [ +export const MOBILE_MARKDOWN_TAB_METHODS = [ defineMethod({ name: 'markdown.readTab', params: ActivateTab, diff --git a/src/main/runtime/rpc/methods/native-chat.ts b/src/main/runtime/rpc/methods/native-chat.ts index 305e8adb771..05f8f7f8726 100644 --- a/src/main/runtime/rpc/methods/native-chat.ts +++ b/src/main/runtime/rpc/methods/native-chat.ts @@ -5,7 +5,7 @@ import { type NativeChatTranscriptSubscription, type SubscribeNativeChatTranscriptArgs } from '../../../native-chat/transcript-watch' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, defineStreamingMethod, type RpcContext } from '../core' import { sanitizeNativeChatRpcBlock } from './native-chat-rpc-block-sanitize' import { MOBILE_NATIVE_CHAT_MAX_WINDOW, @@ -63,7 +63,7 @@ function windowForClient( return windowed.map((message) => sanitizeMessage(message, clientKind)) } -export const NATIVE_CHAT_METHODS: readonly RpcAnyMethod[] = [ +export const NATIVE_CHAT_METHODS = [ defineMethod({ name: 'nativeChat.readSession', params: NativeChatSession, diff --git a/src/main/runtime/rpc/methods/notifications.ts b/src/main/runtime/rpc/methods/notifications.ts index 2b9e05fb6f5..8168da70cae 100644 --- a/src/main/runtime/rpc/methods/notifications.ts +++ b/src/main/runtime/rpc/methods/notifications.ts @@ -1,5 +1,5 @@ import { createNotificationStreamFilter } from './notification-stream-policy' -import { defineStreamingMethod, defineMethod, type RpcAnyMethod } from '../core' +import { defineStreamingMethod, defineMethod } from '../core' import { NotificationGetMissedSinceParams, NotificationRegisterPushParams, @@ -13,7 +13,7 @@ import { let notificationsSubscriptionSeq = 0 // Legacy callers retain filtered socket alerts; push clients opt into the full event stream. -export const NOTIFICATION_METHODS: readonly RpcAnyMethod[] = [ +export const NOTIFICATION_METHODS = [ defineStreamingMethod({ name: 'notifications.subscribe', params: NotificationsSubscribeParams, diff --git a/src/main/runtime/rpc/methods/orchestration.ts b/src/main/runtime/rpc/methods/orchestration.ts index ed89ae4519d..ab80b91e830 100644 --- a/src/main/runtime/rpc/methods/orchestration.ts +++ b/src/main/runtime/rpc/methods/orchestration.ts @@ -1,4 +1,3 @@ -import type { RpcMethod } from '../core' import { ORCHESTRATION_RUN_METHODS } from './orchestration/runs/runs' import { ORCHESTRATION_WORKER_METHODS } from './orchestration/worker/worker-methods' import { ORCHESTRATION_FEDERATION_METHODS } from './orchestration/federation/federation-methods' @@ -11,7 +10,7 @@ import { ORCHESTRATION_ASK_METHODS } from './orchestration/messaging/ask-methods import { ORCHESTRATION_GATE_METHODS } from './orchestration/gates/gates' import { ORCHESTRATION_RESET_METHODS } from './orchestration/runs/reset-methods' -export const ORCHESTRATION_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_METHODS = [ ...ORCHESTRATION_RUN_METHODS, ...ORCHESTRATION_WORKER_METHODS, ...ORCHESTRATION_FEDERATION_METHODS, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts index d5150dc052e..87ac8f65356 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts @@ -3,6 +3,7 @@ import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../../../shared/protoco import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' const HOME_FINGERPRINT = 'home-peer' const PANE_KEY = 'tab_remote:bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' @@ -191,7 +192,9 @@ describe('federated worker release ownership', () => { dispatchId: string, params: Record = { dispatchId } ): Promise { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts index 0ab3cce34aa..caec51d462e 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts @@ -1,6 +1,6 @@ import { OrchestrationError } from '../../../../orchestration/orchestration-error' import type { RemoteDispatchAttachmentRow } from '../../../../orchestration/types' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { mapWithConcurrency } from '../../../../../../shared/map-with-concurrency' import { readExactWorkerOutput } from '../worker/worker-output' import { describeUnconfirmedAgentStop } from '../../../../../../shared/pty-liveness-verdict' @@ -16,7 +16,7 @@ import { FederationReadParams } from '../../../../../../shared/rpc-contract/orchestration-federation-control-params' -export const ORCHESTRATION_FEDERATION_CONTROL_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_CONTROL_METHODS = [ defineMethod({ name: 'orchestration.federationFleetSnapshot', params: FederationFleetSnapshotParams, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts index 7c75c52eb6c..dc5989fff2a 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts @@ -4,6 +4,7 @@ import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../../../shared/protoco import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' // The federation host runs its own copy of the observation and stop logic, so // it needs the same rule: lost contact with a worker's host is not an exit, and @@ -87,7 +88,9 @@ describe('federation host liveness verdicts', () => { afterEach(() => db.close()) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } @@ -138,7 +141,9 @@ describe('federation host liveness verdicts', () => { }) hostDb.markRemoteAttachmentReady(DISPATCH_ID) const callHost = async (name: string, params: Record) => { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts index fbdbda6c7ca..589b69f3934 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts @@ -1,9 +1,8 @@ -import type { RpcMethod } from '../../../core' import { ORCHESTRATION_FEDERATION_CONTROL_METHODS } from './federation-control' import { ORCHESTRATION_FEDERATION_RELAY_METHODS } from './federation-relay' import { ORCHESTRATION_FEDERATION_ATTACH_METHODS } from './federation' -export const ORCHESTRATION_FEDERATION_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_METHODS = [ ...ORCHESTRATION_FEDERATION_ATTACH_METHODS, ...ORCHESTRATION_FEDERATION_RELAY_METHODS, ...ORCHESTRATION_FEDERATION_CONTROL_METHODS diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts index 58561e5e044..721232f194a 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts @@ -13,7 +13,7 @@ import { FederationPullParams } from '../../../../../../shared/rpc-contract/orchestration-federation-relay-params' -export const ORCHESTRATION_FEDERATION_RELAY_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_RELAY_METHODS = [ defineMethod({ name: 'orchestration.federationPull', params: FederationPullParams, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation.ts index 785f6a67eec..dbf6a5f4b5e 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation.ts @@ -1,7 +1,7 @@ import type { TuiAgent } from '../../../../../../shared/tui-agent' import { buildDispatchPreamble } from '../../../../orchestration/preamble' import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { assertOrchestrationWorktreeCreationSupported } from '../worker/folder-worktree-placement' import { appendFederationSetupEffect, @@ -24,7 +24,7 @@ import { } from '../../../../../../shared/orchestration-timing-budgets' import { assertWorkerStartTaskSpecWithinPromptBudget } from '../worker/worker-start-prompt-budget' -export const ORCHESTRATION_FEDERATION_ATTACH_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_ATTACH_METHODS = [ defineMethod({ name: 'orchestration.federationAttachStart', params: FederationAttachStartParams, diff --git a/src/main/runtime/rpc/methods/orchestration/gates/gates.ts b/src/main/runtime/rpc/methods/orchestration/gates/gates.ts index 6bb8750628e..29be91b4324 100644 --- a/src/main/runtime/rpc/methods/orchestration/gates/gates.ts +++ b/src/main/runtime/rpc/methods/orchestration/gates/gates.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import type { GateStatus } from '../../../../orchestration/db' import { Coordinator } from '../../../../orchestration/coordinator' import { resolveRunScope } from '../runs/run-scope' @@ -16,7 +16,7 @@ import { // the DB's active-run check), so a single reference suffices. let activeCoordinator: Coordinator | null = null -export const ORCHESTRATION_GATE_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_GATE_METHODS = [ // Why: Section 4.12 — orchestration.run returns immediately with a run ID. // The coordinator loop runs in the background; progress is queried via // orchestration.taskList. This prevents the RPC call from blocking the diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts index f622cc9e67a..ef191250a8b 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { clampOrchestrationAskTimeoutMs } from '../../../../../../shared/orchestration-ask-timeout' import { isGroupAddress } from '../../../../orchestration/groups' @@ -6,7 +6,7 @@ import { AskParams } from '../schemas' import { rejectFederatedExplicitTarget } from '../routing' import { askRemoteRunHome } from './ask-remote' -export const ORCHESTRATION_ASK_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_ASK_METHODS = [ defineMethod({ name: 'orchestration.ask', params: AskParams, diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts index a12428253c3..20ac25e6512 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { CheckParams } from '../schemas' import { parseMessageTypes } from '../routing' @@ -12,7 +12,7 @@ import { isSupersededDispatch } from './dispatch-mailbox-fence' -export const ORCHESTRATION_CHECK_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_CHECK_METHODS = [ defineMethod({ name: 'orchestration.check', params: CheckParams, diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts index 58e0b3aa0ca..f0d5e4933a6 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts @@ -3,7 +3,7 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' import { ORCHESTRATION_METHODS } from '../../orchestration' -import type { RpcContext } from '../../../core' +import { eraseRpcMethods, type RpcContext } from '../../../core' import { OrchestrationDb } from '../../../../orchestration/db' import { OrcaRuntimeService } from '../../../../orca-runtime' import { @@ -55,7 +55,9 @@ describe('orchestration.check on a federated attachment across a restart', () => } function check(ctx: RpcContext, params: Record = {}): Promise { - const method = ORCHESTRATION_METHODS.find((entry) => entry.name === 'orchestration.check') + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (entry) => entry.name === 'orchestration.check' + ) if (!method) { throw new Error('orchestration.check is not registered') } diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts index 61808c19aed..51a9d9bc0c5 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import type { TaskStatus } from '../../../../orchestration/db' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { ORCHESTRATION_LEGACY_RUN_ID } from '../../../../../../shared/orchestration-rpc-contract' @@ -19,7 +19,7 @@ import { TaskUpdateParams } from '../schemas' -export const ORCHESTRATION_MESSAGE_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_MESSAGE_METHODS = [ defineMethod({ name: 'orchestration.reply', params: ReplyParams, diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts index 567149e69d8..7d1edb77602 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { isGroupAddress } from '../../../../orchestration/groups' import { orchestrationSkillRecoveryData } from '../../../../../../shared/orchestration-rpc-contract' @@ -20,7 +20,7 @@ import { sendPointToPointMessage } from './send-point-to-point' import { sendGroupMessage } from './send-group' import { sendFederatedControlMail } from './send-control-mail' -export const ORCHESTRATION_SEND_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_SEND_METHODS = [ defineMethod({ name: 'orchestration.send', params: SendParams, diff --git a/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts b/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts index dfba4bd143f..19e636eca21 100644 --- a/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts +++ b/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts @@ -1,6 +1,6 @@ import { vi } from 'vitest' import { ORCHESTRATION_METHODS } from '../orchestration' -import type { RpcContext } from '../../core' +import { eraseRpcMethods, type RpcContext } from '../../core' import { OrchestrationDb } from '../../../orchestration/db' import { OrcaRuntimeService } from '../../../orca-runtime' @@ -66,7 +66,7 @@ export function createOrchestrationRpcHarness() { } function findMethod(name: string) { - const method = ORCHESTRATION_METHODS.find((m) => m.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find((m) => m.name === name) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts b/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts index abc941bf98c..a4e6b427d7d 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { buildDispatchPreamble } from '../../../../orchestration/preamble' import { resolveDispatchCreator } from './dispatch-creator' @@ -10,7 +10,7 @@ import { import { resolveRunScope } from './run-scope' import { DispatchParams, DispatchShowParams } from '../schemas' -export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_DISPATCH_METHODS = [ defineMethod({ name: 'orchestration.dispatch', params: DispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts b/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts index 8ca5333f307..a1cf43ab424 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts @@ -2,10 +2,10 @@ import { describeMutationRequestState, type OrchestrationMutationRequestShowResult } from '../../../../../../shared/orchestration-mutation-request' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { RequestShowParams } from '../../../../../../shared/rpc-contract/orchestration-runs-mutation-request-show-params' -export const ORCHESTRATION_MUTATION_REQUEST_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_MUTATION_REQUEST_METHODS = [ defineMethod({ name: 'orchestration.requestShow', params: RequestShowParams, diff --git a/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts b/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts index b4be53ecad5..6946606651f 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { ResetParams } from '../schemas' -export const ORCHESTRATION_RESET_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_RESET_METHODS = [ defineMethod({ name: 'orchestration.reset', params: ResetParams, diff --git a/src/main/runtime/rpc/methods/orchestration/runs/runs.ts b/src/main/runtime/rpc/methods/orchestration/runs/runs.ts index dc0b112a168..eab6eb2913e 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/runs.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/runs.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { assertCallerHandleMatchesEvidence, resolveOrchestrationCaller } from './run-scope' import { exposeRun } from './run-receipt' @@ -10,7 +10,7 @@ import { RunUseParams } from '../../../../../../shared/rpc-contract/orchestration-runs-params' -export const ORCHESTRATION_RUN_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_RUN_METHODS = [ defineMethod({ name: 'orchestration.runCreate', params: RunCreateParams, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts index 2890fa08938..25da0b311a3 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts @@ -3,6 +3,7 @@ import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' import { createRootDispatch } from '../../../../orchestration/db/root-dispatch-test-fixture' +import { eraseRpcMethods } from '../../../core' describe('manual Dispatch observation', () => { let db: OrchestrationDb | undefined @@ -51,7 +52,7 @@ describe('manual Dispatch observation', () => { coordinatorPaneKey }) const task = db.createTask({ spec: 'injected lane', runId: run.id }) - const dispatchMethod = ORCHESTRATION_METHODS.find( + const dispatchMethod = eraseRpcMethods(ORCHESTRATION_METHODS).find( (candidate) => candidate.name === 'orchestration.dispatch' ) if (!dispatchMethod) { @@ -77,7 +78,7 @@ describe('manual Dispatch observation', () => { capability_hash: expect.any(String) }) - const workerShowMethod = ORCHESTRATION_METHODS.find( + const workerShowMethod = eraseRpcMethods(ORCHESTRATION_METHODS).find( (candidate) => candidate.name === 'orchestration.workerShow' ) if (!workerShowMethod) { @@ -132,7 +133,9 @@ describe('manual Dispatch observation', () => { }) const context = { runtime } const call = async (name: string, params: Record) => { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Missing method ${name}`) } @@ -233,7 +236,7 @@ describe('manual Dispatch observation', () => { const task = db.createTask({ spec: 'operator lane', runId: run.id }) const dispatch = createRootDispatch(db, task.id, 'term_worker', 'tab_worker:leaf_worker') - const workerListMethod = ORCHESTRATION_METHODS.find( + const workerListMethod = eraseRpcMethods(ORCHESTRATION_METHODS).find( (candidate) => candidate.name === 'orchestration.workerList' ) if (!workerListMethod) { @@ -280,7 +283,9 @@ describe('manual Dispatch observation', () => { 'launch-hash', 'runtime_test:term_worker:1' ) - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Missing method ${name}`) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts index ee1f5a3162a..e340e51b01d 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts @@ -3,6 +3,7 @@ import type Database from '../../../../../sqlite/sync-database' import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' const COORDINATOR = 'term_coordinator' const TARGET = 'term_target' @@ -177,7 +178,9 @@ describe('manual Dispatch release', () => { } async function call(name: string, params: Record): Promise { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts index 3e27e12eba0..5e8babf3c5e 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts @@ -1,6 +1,6 @@ import { contextOnlyAbandonWarning } from '../../../../orchestration/context-only-dispatch-release' import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { exposeDispatchContext, exposeObservation, @@ -22,7 +22,7 @@ import { WorkerReadParams } from '../../../../../../shared/rpc-contract/orchestration-worker-control-params' -export const ORCHESTRATION_WORKER_CONTROL_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_CONTROL_METHODS = [ defineMethod({ name: 'orchestration.workerShow', params: WorkerDispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts index 5c8ae65fa86..c0f8097a690 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts @@ -4,7 +4,7 @@ import type { OrchestrationDb } from '../../../../orchestration/db' import { WORKER_LIST_CURSOR_EXPIRED_MESSAGE } from '../../../../orchestration/db/worker-terminal/worker-terminal-listing' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import type { OrcaRuntimeService } from '../../../../orca-runtime' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { applyFederatedFleetObservations, readFederatedFleetSnapshots @@ -24,7 +24,7 @@ import { projectWorkerFleet, type WorkerListPageParams } from './worker-list-pro import { exposeWorkerTerminalResource } from './worker-release-completion' import { WORKER_TERMINAL_LIST_STATES, WorkerListParams } from './worker-release-schemas' -export const ORCHESTRATION_WORKER_LIST_METHOD: RpcMethod = defineMethod({ +export const ORCHESTRATION_WORKER_LIST_METHOD = defineMethod({ name: 'orchestration.workerList', params: WorkerListParams, handler: async (params, { runtime }) => { diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts index 238ad12fad8..7c324cdf798 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts @@ -1,10 +1,9 @@ -import type { RpcMethod } from '../../../core' import { ORCHESTRATION_WORKER_CONTROL_METHODS } from './worker-control' import { ORCHESTRATION_WORKER_RELEASE_METHODS } from './worker-release' import { ORCHESTRATION_WORKER_STOP_METHODS } from './worker-stop' import { ORCHESTRATION_WORKER_START_METHODS } from './workers' -export const ORCHESTRATION_WORKER_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_METHODS = [ ...ORCHESTRATION_WORKER_START_METHODS, ...ORCHESTRATION_WORKER_CONTROL_METHODS, ...ORCHESTRATION_WORKER_STOP_METHODS, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts index 68d40b4a04e..dd4ce2522ea 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts @@ -2,7 +2,7 @@ import { afterEach, beforeEach, expect, it, vi } from 'vitest' import { createOrchestrationWorkerReleaseHarness } from './worker-release.test-support' import { TERMINAL_SEND_METHODS } from '../../terminal/terminal-send-method' import { sendTerminalStreamInput } from '../../terminal/terminal-input-delivery' -import { isStreamingMethod, type RpcMethod } from '../../../core' +import { eraseRpcMethods, isStreamingMethod, type RpcMethod } from '../../../core' const h = createOrchestrationWorkerReleaseHarness() beforeEach(() => h.setup()) @@ -93,7 +93,7 @@ it.each(['unary', 'stream'])('mobile %s bytes do no orchestration database work' 'delivered' ) } else { - const method = TERMINAL_SEND_METHODS.find( + const method = eraseRpcMethods(TERMINAL_SEND_METHODS).find( (m): m is RpcMethod => m.name === 'terminal.send' && !isStreamingMethod(m) )! await expect( diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts index a7481ea3b68..177eb479d42 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts @@ -2,7 +2,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { OrchestrationDb } from '../../../../orchestration/db' import { reconcileRequestedWorkerTerminalReleases } from '../../../../orchestration/worker-terminal-release-reconciliation' import { OrcaRuntimeService } from '../../../../orca-runtime' -import type { RpcContext } from '../../../core' +import { eraseRpcMethods, type RpcContext } from '../../../core' import { ORCHESTRATION_METHODS } from '../../orchestration' function deferred(): { promise: Promise; resolve: (value: T) => void } { @@ -101,7 +101,9 @@ describe('orchestration worker release recovery', () => { }) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts index ff1ea59a263..a13ea320670 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts @@ -1,6 +1,6 @@ import { expect, vi } from 'vitest' import { ORCHESTRATION_METHODS } from '../../orchestration' -import type { RpcContext } from '../../../core' +import { eraseRpcMethods, type RpcContext } from '../../../core' import { OrchestrationDb } from '../../../../orchestration/db' import { OrcaRuntimeService } from '../../../../orca-runtime' @@ -130,7 +130,7 @@ export function createOrchestrationWorkerReleaseHarness(): OrchestrationWorkerRe } function findMethod(name: string) { - const method = ORCHESTRATION_METHODS.find((m) => m.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find((m) => m.name === name) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts index 236dc7cf76f..f641485e757 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts @@ -1,5 +1,5 @@ import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { releaseFederatedWorker } from '../federation/federated-worker-release' import { ORCHESTRATION_WORKER_LIST_METHOD } from './worker-list-method' import { resolvePinnedFederatedServer } from './worker-observation' @@ -11,7 +11,7 @@ import { import { WorkerDispatchParams, WorkerRetainParams } from './worker-release-schemas' import { OrchestrationWorkerTerminalUserInputParams } from '../../../../../../shared/rpc-contract/orchestration-worker-release-params' -export const ORCHESTRATION_WORKER_RELEASE_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_RELEASE_METHODS = [ defineMethod({ name: 'orchestration.workerRelease', params: WorkerDispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts index f0b65281df1..c8dec854e4c 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' // The aggregate terminal inventory only iterates registered providers, so a // dropped relay clears `connected` for every remote PTY at once. That is lost @@ -29,7 +30,9 @@ describe('worker-stop against a terminal we lost contact with', () => { afterEach(() => db.close()) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts index 79a51ca506b..98d3c376f61 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts @@ -1,5 +1,5 @@ import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { describeUnconfirmedAgentStop } from '../../../../../../shared/pty-liveness-verdict' import { ORCHESTRATION_WORKER_STOP_VERDICT_RUNTIME_CAPABILITY } from '../../../../../../shared/protocol-version' import type { RuntimeStatus } from '../../../../../../shared/runtime-types' @@ -12,7 +12,7 @@ import { import { isStructuredWorkerHandle } from '../../../../structured-worker-identity' import { WorkerDispatchParams } from '../../../../../../shared/rpc-contract/orchestration-worker-stop-params' -export const ORCHESTRATION_WORKER_STOP_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_STOP_METHODS = [ defineMethod({ name: 'orchestration.workerStop', params: WorkerDispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts index a635a316b23..9e95d7f33e2 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' function deferred(): { promise: Promise; resolve: (value: T) => void } { let resolve!: (value: T) => void @@ -46,7 +47,9 @@ describe('orchestration worker recovery', () => { afterEach(() => db.close()) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/workers.ts b/src/main/runtime/rpc/methods/orchestration/worker/workers.ts index 8b14ec044cf..b1a1f40d45a 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/workers.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/workers.ts @@ -1,5 +1,5 @@ import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { startFederatedWorker } from '../federation/federated-worker-start' import { startLocalWorker } from './local-worker-start' import { @@ -14,7 +14,7 @@ import { } from '../../../../../../shared/orchestration-timing-budgets' import { assertWorkerStartTaskSpecWithinPromptBudget } from './worker-start-prompt-budget' -export const ORCHESTRATION_WORKER_START_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_START_METHODS = [ defineMethod({ name: 'orchestration.workerStart', params: WorkerStartParams, diff --git a/src/main/runtime/rpc/methods/pairing.ts b/src/main/runtime/rpc/methods/pairing.ts index 7762881ba37..5a32ddab62f 100644 --- a/src/main/runtime/rpc/methods/pairing.ts +++ b/src/main/runtime/rpc/methods/pairing.ts @@ -1,10 +1,10 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { PairingGetEndpointsParamsSchema, PairingProvisionRelayParamsSchema } from '../../../../shared/mobile-relay-credential-contract' -export const PAIRING_METHODS: readonly RpcAnyMethod[] = [ +export const PAIRING_METHODS = [ defineMethod({ name: 'pairing.getEndpoints', params: PairingGetEndpointsParamsSchema, diff --git a/src/main/runtime/rpc/methods/plugins.test.ts b/src/main/runtime/rpc/methods/plugins.test.ts index bf67d29f19a..e44570bab56 100644 --- a/src/main/runtime/rpc/methods/plugins.test.ts +++ b/src/main/runtime/rpc/methods/plugins.test.ts @@ -1,12 +1,12 @@ import { afterEach, describe, expect, it, vi } from 'vitest' -import type { RpcContext, RpcMethod } from '../core' +import { eraseRpcMethods, type RpcContext, type RpcMethod } from '../core' import type { PluginService } from '../../../plugins/plugin-service' import { PLUGIN_METHODS, setPluginServiceForRpc } from './plugins' const SESSION_TOKEN = 's'.repeat(43) function method(name: string): RpcMethod { - const found = PLUGIN_METHODS.find((entry) => entry.name === name) + const found = eraseRpcMethods(PLUGIN_METHODS).find((entry) => entry.name === name) if (!found) { throw new Error(`missing ${name}`) } diff --git a/src/main/runtime/rpc/methods/plugins.ts b/src/main/runtime/rpc/methods/plugins.ts index 4d1e8d597ca..667aff9179d 100644 --- a/src/main/runtime/rpc/methods/plugins.ts +++ b/src/main/runtime/rpc/methods/plugins.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcContext, type RpcMethod } from '../core' +import { defineMethod, type RpcContext } from '../core' import type { PluginPanelEntry } from '../../../../shared/plugins/plugin-panel-bridge' import { listPluginsForClients } from '../../../plugins/plugin-client-list' import type { PluginListEntry } from '../../../plugins/plugin-list-projection' @@ -65,7 +65,7 @@ function bindRpcPanelOwner(service: PluginService, context: RpcContext): string return ownerKey } -export const PLUGIN_METHODS: readonly RpcMethod[] = [ +export const PLUGIN_METHODS = [ defineMethod({ name: 'plugins.list', params: null, diff --git a/src/main/runtime/rpc/methods/preflight.ts b/src/main/runtime/rpc/methods/preflight.ts index 9a5af8776f1..cc1dd5705c3 100644 --- a/src/main/runtime/rpc/methods/preflight.ts +++ b/src/main/runtime/rpc/methods/preflight.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { detectRemoteAgents, detectRemoteWindowsTerminalCapabilities, @@ -12,7 +12,7 @@ import { PreflightDetectRemoteWindowsTerminalCapabilities } from '../../../../shared/rpc-contract/preflight-params' -export const PREFLIGHT_METHODS: RpcMethod[] = [ +export const PREFLIGHT_METHODS = [ defineMethod({ name: 'preflight.check', params: PreflightCheck, diff --git a/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts b/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts index 2ef04798813..f67705f6cdd 100644 --- a/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts +++ b/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { projectRepoResultVisibilityForClient } from '../repo-visibility-projection' import { ProjectHostSetupClone, @@ -9,7 +9,7 @@ import { ProjectUpdate } from '../../../../shared/rpc-contract/project-runtime-params' -export const PROJECT_RUNTIME_METHODS: RpcMethod[] = [ +export const PROJECT_RUNTIME_METHODS = [ defineMethod({ name: 'project.list', params: null, diff --git a/src/main/runtime/rpc/methods/repo.ts b/src/main/runtime/rpc/methods/repo.ts index 31fc871e897..6498de8a4f3 100644 --- a/src/main/runtime/rpc/methods/repo.ts +++ b/src/main/runtime/rpc/methods/repo.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { PROJECT_RUNTIME_METHODS } from './project-runtime-rpc-methods' import { FOLDER_WORKSPACE_METHODS } from './folder-workspace' import { RepoSelector } from './github-repo-target-schemas' @@ -24,7 +24,7 @@ import { RepoUpdate } from '../../../../shared/rpc-contract/repo-params' -export const REPO_METHODS: RpcMethod[] = [ +export const REPO_METHODS = [ defineMethod({ name: 'repo.list', params: null, diff --git a/src/main/runtime/rpc/methods/runtime-client-capabilities.ts b/src/main/runtime/rpc/methods/runtime-client-capabilities.ts index fe152fa6f27..2fa62b53934 100644 --- a/src/main/runtime/rpc/methods/runtime-client-capabilities.ts +++ b/src/main/runtime/rpc/methods/runtime-client-capabilities.ts @@ -1,8 +1,8 @@ import type { RuntimeCapability } from '../../../../shared/protocol-version' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ClientCapabilitiesUpdate } from '../../../../shared/rpc-contract/runtime-client-capabilities-params' -export const RUNTIME_CLIENT_CAPABILITY_METHODS: RpcAnyMethod[] = [ +export const RUNTIME_CLIENT_CAPABILITY_METHODS = [ defineMethod({ name: 'runtime.clientCapabilities.update', params: ClientCapabilitiesUpdate, diff --git a/src/main/runtime/rpc/methods/session-tab-close-methods.ts b/src/main/runtime/rpc/methods/session-tab-close-methods.ts index 4800b7d33c1..a7065cf6ba2 100644 --- a/src/main/runtime/rpc/methods/session-tab-close-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-close-methods.ts @@ -1,13 +1,13 @@ import { withSpan } from '../../../observability/tracer' import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { CloseLifecycleTab, CloseTab } from './session-tabs-schemas' import { assertProjectedSessionTabVisible } from './session-tab-browser-placement-projection' import { assertAgentSessionTabDestructiveMutationSupported } from './session-tab-agent-status-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' -export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_CLOSE_METHODS = [ defineMethod({ name: 'session.tabs.close', params: CloseTab, diff --git a/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts b/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts index 6144cd3e546..f2be1d4a61d 100644 --- a/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' -export const SESSION_TAB_MARKDOWN_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_MARKDOWN_METHODS = [ defineMethod({ name: 'markdown.readTab', params: ActivateTab, diff --git a/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts b/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts index 462d00d869d..d62f50be595 100644 --- a/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts @@ -1,6 +1,6 @@ import { resolveRuntimeNavigationTarget } from '../../../../shared/runtime-navigation' import type { OrcaRuntimeService } from '../../orca-runtime' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { assertProjectedSessionTabVisible, translateProjectedSessionTabMove @@ -9,7 +9,7 @@ import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' import { ActivateTab, MoveTab, SetTabProps, UpdatePaneLayout } from './session-tabs-schemas' -export const SESSION_TAB_MUTATION_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_MUTATION_METHODS = [ defineMethod({ name: 'session.tabs.activate', params: ActivateTab, diff --git a/src/main/runtime/rpc/methods/session-tabs.ts b/src/main/runtime/rpc/methods/session-tabs.ts index 1f4019bad7c..d6441ee84cb 100644 --- a/src/main/runtime/rpc/methods/session-tabs.ts +++ b/src/main/runtime/rpc/methods/session-tabs.ts @@ -1,5 +1,5 @@ import { resolveRuntimeNavigationTarget } from '../../../../shared/runtime-navigation' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { CreateTerminalTab, SessionTabsUnsubscribe, @@ -19,7 +19,7 @@ import { isStructuredNativeChatEnabled } from './structured-agent-session-policy import { assertLegacyAiVaultResumeCommandAllowed } from '../../../ai-vault/structured-session-ownership' import { SessionTabsUnsubscribeAllParams } from '../../../../shared/rpc-contract/session-tabs-params' -export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_METHODS = [ defineMethod({ name: 'session.tabs.list', params: WorktreeTabSelector, diff --git a/src/main/runtime/rpc/methods/skills.test.ts b/src/main/runtime/rpc/methods/skills.test.ts index 0425e5922eb..9bc5a647c82 100644 --- a/src/main/runtime/rpc/methods/skills.test.ts +++ b/src/main/runtime/rpc/methods/skills.test.ts @@ -1,5 +1,5 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' -import type { RpcContext } from '../core' +import { eraseRpcMethods, type RpcContext } from '../core' vi.mock('electron', () => ({ app: { getPath: () => '/orca-state', isPackaged: true } @@ -41,7 +41,7 @@ function makeContext(overrides: { } function discoverMethod() { - const method = SKILL_METHODS.find((entry) => entry.name === 'skills.discover') + const method = eraseRpcMethods(SKILL_METHODS).find((entry) => entry.name === 'skills.discover') if (!method) { throw new Error('skills.discover method not registered') } @@ -49,7 +49,7 @@ function discoverMethod() { } function installMethod() { - const method = SKILL_METHODS.find((entry) => entry.name === 'skills.install') + const method = eraseRpcMethods(SKILL_METHODS).find((entry) => entry.name === 'skills.install') if (!method) { throw new Error('skills.install method not registered') } @@ -57,7 +57,7 @@ function installMethod() { } function method(name: string) { - const value = SKILL_METHODS.find((entry) => entry.name === name) + const value = eraseRpcMethods(SKILL_METHODS).find((entry) => entry.name === name) if (!value) { throw new Error(`${name} method not registered`) } diff --git a/src/main/runtime/rpc/methods/skills.ts b/src/main/runtime/rpc/methods/skills.ts index 01caa64baa6..39a3a73eb7c 100644 --- a/src/main/runtime/rpc/methods/skills.ts +++ b/src/main/runtime/rpc/methods/skills.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import type { z } from 'zod' import { getAppEnvironment } from '../../../../shared/app-environment' import { SkillDeleteRequestSchema } from '../../../../shared/skill-delete-contract' @@ -64,7 +64,7 @@ function skillDeleteDependencies( } } -export const SKILL_METHODS: RpcMethod[] = [ +export const SKILL_METHODS = [ defineMethod({ name: 'skills.discover', params: SkillsDiscoverParams, diff --git a/src/main/runtime/rpc/methods/speech.ts b/src/main/runtime/rpc/methods/speech.ts index 8e5e8bdbd4d..086a528ab0e 100644 --- a/src/main/runtime/rpc/methods/speech.ts +++ b/src/main/runtime/rpc/methods/speech.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { DictationChunk, DictationHandle, @@ -7,7 +7,7 @@ import { SpeechModelAction } from '../../../../shared/rpc-contract/speech-params' -export const SPEECH_METHODS: RpcMethod[] = [ +export const SPEECH_METHODS = [ defineMethod({ name: 'speech.models.list', params: null, diff --git a/src/main/runtime/rpc/methods/ssh.ts b/src/main/runtime/rpc/methods/ssh.ts index 8988ab3a569..2c8e2520f9b 100644 --- a/src/main/runtime/rpc/methods/ssh.ts +++ b/src/main/runtime/rpc/methods/ssh.ts @@ -4,7 +4,7 @@ import { listRegisteredRemovedSshTargetLabels, listRegisteredSshTargets } from '../../../ssh/ssh-target-registry' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { getPublicSshError, getPublicSshState } from '../../public-ssh-state' import type { SshTargetSummary } from '../../../../shared/ssh-types' import { SshTarget } from '../../../../shared/rpc-contract/ssh-params' @@ -25,7 +25,7 @@ function listRegisteredSshTargetSummaries(): SshTargetSummary[] { }) } -export const SSH_METHODS: RpcMethod[] = [ +export const SSH_METHODS = [ defineMethod({ name: 'ssh.getState', params: SshTarget, diff --git a/src/main/runtime/rpc/methods/stats.ts b/src/main/runtime/rpc/methods/stats.ts index 59f71701c3a..9cdfe5269d3 100644 --- a/src/main/runtime/rpc/methods/stats.ts +++ b/src/main/runtime/rpc/methods/stats.ts @@ -1,6 +1,6 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' -export const STATS_METHODS: RpcMethod[] = [ +export const STATS_METHODS = [ defineMethod({ name: 'stats.summary', params: null, diff --git a/src/main/runtime/rpc/methods/status.ts b/src/main/runtime/rpc/methods/status.ts index 03d66f84fb1..dac38097b7a 100644 --- a/src/main/runtime/rpc/methods/status.ts +++ b/src/main/runtime/rpc/methods/status.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { getRemoteServerUpdaterSnapshot } from '../../remote-server-updater' -export const STATUS_METHODS: RpcMethod[] = [ +export const STATUS_METHODS = [ defineMethod({ name: 'status.get', params: null, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-hold.ts b/src/main/runtime/rpc/methods/structured-agent-session-hold.ts index 280804711e6..18fb600a949 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-hold.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-hold.ts @@ -9,7 +9,7 @@ // the hold is deliberate: re-registering an id runs the previous cleanup synchronously, so the // stale release lands before this hold rather than after it. -import { defineMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled, requireStructuredCleanupHost, @@ -28,7 +28,7 @@ function holdCleanupIdFor(sessionId: string, holderKey: string): string { return `${HOLD_CLEANUP_PREFIX}:${holderKey}:${sessionId}` } -export const STRUCTURED_AGENT_SESSION_HOLD_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_HOLD_METHODS = [ defineMethod({ name: 'agentSession.hold', params: HoldParams, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts b/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts index 5f2ab0e8cac..47f30a5030d 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts @@ -12,7 +12,7 @@ import { isAgentSessionWireRefusalCode } from '../../../../shared/agent-session-wire' import type { StructuredAgentSessionReveal } from '../../../native-chat/agent-session-wire/structured-agent-session-host-types' import { refuseAgentSessionMutation } from '../../../native-chat/agent-session-wire/structured-agent-session-mutation-admission' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ensureStructuredHostInstalled, requireStructuredCapability, @@ -20,7 +20,7 @@ import { } from './structured-agent-session-gate' import { OptionsParams } from './structured-agent-session-schemas' -export const STRUCTURED_AGENT_SESSION_REVEAL_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_REVEAL_METHODS = [ defineMethod({ name: 'agentSession.reveal', params: OptionsParams, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts b/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts index 8637089c254..c93401e0e22 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts @@ -3,7 +3,7 @@ // Session lists read turn state from here instead of replaying transcripts: one stream per client // covers every session, and unlike a transcript subscription it retains none of them. -import { defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineStreamingMethod, type RpcContext } from '../core' import { requireStructuredHost as requireHost } from './structured-agent-session-gate' import { structuredAgentSessionStatusSubscriptionId } from './structured-agent-session-subscription-id' @@ -41,7 +41,7 @@ export function bindStructuredAgentSessionStream( return { isClosed: () => closed } } -export const STRUCTURED_AGENT_SESSION_STATUS_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_STATUS_METHODS = [ defineStreamingMethod({ name: 'agentSession.subscribeStatus', params: null, diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index c09ae4cfcbf..f1d0fc59ec5 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -18,7 +18,7 @@ import { projectTurnItemEvent, projectTurnItemHistory } from './structured-agent-session-turn-item-capability' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, defineStreamingMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled as ensureHostInstalled, requireStructuredCapability, @@ -91,7 +91,7 @@ async function attachClientSuppliedLocation( return host.attach(callerFor(ctx), attachParams) } -export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_METHODS = [ defineMethod({ name: 'agentSession.rewind', params: RewindParams, diff --git a/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts b/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts index 151285ffb1e..aebb420fa8b 100644 --- a/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts +++ b/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts @@ -16,6 +16,7 @@ import { structuredWorkerProcessIncarnation } from '../../structured-worker-identity' import { ORCHESTRATION_METHODS } from './orchestration' +import { eraseRpcMethods } from '../core' const SESSION = 'session-stop-receipt' const HANDLE = 'structworker_22222222-2222-4222-a222-222222222222' @@ -43,7 +44,9 @@ describe('worker-stop on a structured worker this runtime cannot reach', () => { }) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts b/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts index 51001605be0..a1e221d5a7a 100644 --- a/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts +++ b/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it, vi } from 'vitest' -import type { RpcContext } from '../core' +import { eraseRpcMethods, type RpcContext } from '../core' import { TERMINAL_METHODS } from './terminal' describe('terminal.create RPC idempotency', () => { @@ -15,7 +15,9 @@ describe('terminal.create RPC idempotency', () => { run: (worktree: string | undefined, handle: string | undefined) => Promise ) => run('id:worktree-1', 'term_stable') ) - const method = TERMINAL_METHODS.find((candidate) => candidate.name === 'terminal.create') + const method = eraseRpcMethods(TERMINAL_METHODS).find( + (candidate) => candidate.name === 'terminal.create' + ) if (!method) { throw new Error('terminal.create method missing') } @@ -73,7 +75,9 @@ describe('terminal.create RPC idempotency', () => { run: (worktree: string | undefined, handle: string | undefined) => Promise ) => run('id:worktree-1', undefined) ) - const method = TERMINAL_METHODS.find((candidate) => candidate.name === 'terminal.create') + const method = eraseRpcMethods(TERMINAL_METHODS).find( + (candidate) => candidate.name === 'terminal.create' + ) if (!method) { throw new Error('terminal.create method missing') } @@ -114,7 +118,9 @@ describe('terminal.create RPC idempotency', () => { run: (worktree: string | undefined, handle: string | undefined) => Promise ) => run('id:worktree-1', undefined) ) - const method = TERMINAL_METHODS.find((candidate) => candidate.name === 'terminal.create') + const method = eraseRpcMethods(TERMINAL_METHODS).find( + (candidate) => candidate.name === 'terminal.create' + ) if (!method) { throw new Error('terminal.create method missing') } diff --git a/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts b/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts index ccdcf5fb7b1..e26788d5d46 100644 --- a/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts +++ b/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it, vi } from 'vitest' import type { OrcaRuntimeService } from '../../orca-runtime' import { TERMINAL_METHODS } from './terminal' +import { eraseRpcMethods } from '../core' import { TerminalMultiplexLegacyAckFrame, TerminalMultiplexSourceRangeAckFrame, @@ -50,14 +51,14 @@ const METHOD_CASES: readonly (readonly [string, unknown, boolean])[] = [ ] function schemaFor(name: string) { - const method = TERMINAL_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(TERMINAL_METHODS).find((candidate) => candidate.name === name) if (!method?.params) { throw new Error(`Missing terminal schema: ${name}`) } return method.params } async function invoke(name: string, params: unknown, runtime: Partial) { - const method = TERMINAL_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(TERMINAL_METHODS).find((candidate) => candidate.name === name) if (!method?.params || 'stream' in method) { throw new Error(`Missing unary terminal method: ${name}`) } diff --git a/src/main/runtime/rpc/methods/terminal-orphan.ts b/src/main/runtime/rpc/methods/terminal-orphan.ts index 7ca7cd771b1..0b728629dcb 100644 --- a/src/main/runtime/rpc/methods/terminal-orphan.ts +++ b/src/main/runtime/rpc/methods/terminal-orphan.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { TerminalAdoptOrphans } from '../../../../shared/rpc-contract/terminal-orphan-params' -export const TERMINAL_ORPHAN_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_ORPHAN_METHODS = [ defineMethod({ name: 'terminal.adoptOrphans', params: TerminalAdoptOrphans, diff --git a/src/main/runtime/rpc/methods/terminal.ts b/src/main/runtime/rpc/methods/terminal.ts index e80d07773dc..607a29329cd 100644 --- a/src/main/runtime/rpc/methods/terminal.ts +++ b/src/main/runtime/rpc/methods/terminal.ts @@ -1,4 +1,3 @@ -import type { RpcAnyMethod } from '../core' import { TERMINAL_LIFECYCLE_METHODS } from './terminal/terminal-lifecycle-methods' import { TERMINAL_MULTIPLEX_METHODS } from './terminal/terminal-multiplex-method' import { TERMINAL_QUERY_METHODS } from './terminal/terminal-query-methods' @@ -11,7 +10,7 @@ import { // The manifest order is part of the released RPC contract. Keep composition here so the // public entry point owns registration rather than forwarding an aggregated child export. -export const TERMINAL_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_METHODS = [ ...TERMINAL_QUERY_METHODS, ...TERMINAL_SEND_METHODS, ...TERMINAL_LIFECYCLE_METHODS, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts index 48649bc372c..8377f923ecc 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts @@ -6,10 +6,13 @@ import { describe, expect, it, vi } from 'vitest' import type { ZodType } from 'zod' import { TERMINAL_QUERY_METHODS } from './terminal-query-methods' import { TerminalHandle, TerminalInspectProcess } from './unary-schemas' +import { eraseRpcMethods } from '../../core' /** The method as registered, so a schema swap on the definition cannot pass unseen. */ function inspectProcessMethod() { - const method = TERMINAL_QUERY_METHODS.find((entry) => entry.name === 'terminal.inspectProcess') + const method = eraseRpcMethods(TERMINAL_QUERY_METHODS).find( + (entry) => entry.name === 'terminal.inspectProcess' + ) if (!method) { throw new Error('terminal.inspectProcess is not registered') } @@ -25,7 +28,7 @@ async function callRegisteredHandler( foregroundProcess: null, hasChildProcesses: false })) - await method.handler(parsed, { runtime: { inspectTerminalProcess } } as never, undefined as never) + await method.handler(parsed, { runtime: { inspectTerminalProcess } } as never) const [terminal, options] = inspectTerminalProcess.mock.calls[0] as unknown as [string, unknown] return { terminal, options } } diff --git a/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts index 51dde7df4d8..891ed65a13f 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../../core' +import { defineMethod } from '../../core' import { navigationTargetsHost, resolveRuntimeNavigationTarget @@ -19,7 +19,7 @@ import { } from './unary-schemas' import { TerminalResizeForClient } from './stream-schemas' -export const TERMINAL_LIFECYCLE_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_LIFECYCLE_METHODS = [ defineMethod({ name: 'terminal.wait', params: TerminalWait, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts index 11fd0c4d039..e811614e400 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts @@ -1,4 +1,4 @@ -import { defineStreamingMethod, type RpcAnyMethod } from '../../core' +import { defineStreamingMethod } from '../../core' import { TerminalStreamOpcode } from '../../../../../shared/terminal-stream-protocol' import { TERMINAL_MULTIPLEX_ACK_TOTAL_INITIAL_WINDOW_BYTES } from '../../../../../shared/terminal-multiplex-flow-control' import { TerminalSourceRangeRegistry } from '../../terminal-source-range-registry' @@ -11,7 +11,7 @@ import { installMultiplexCleanup } from './terminal-multiplex-cleanup' import { installMultiplexSlotFrames } from './terminal-multiplex-slot-frames' import { installMultiplexSubscribeFrame } from './terminal-multiplex-subscribe-frame' -export const TERMINAL_MULTIPLEX_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_MULTIPLEX_METHODS = [ defineStreamingMethod({ name: 'terminal.multiplex', params: TerminalMultiplex, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts index 82edd55cd79..52c1063b0cc 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../../core' +import { defineMethod } from '../../core' import { TerminalHandle, TerminalInspectProcess, @@ -10,7 +10,7 @@ import { TerminalResolvePane } from './unary-schemas' -export const TERMINAL_QUERY_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_QUERY_METHODS = [ defineMethod({ name: 'terminal.list', params: TerminalListParams, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts index ad471098e49..c62c70c5905 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts @@ -1,6 +1,6 @@ import { isAgentSessionPtyWriteRefusedError } from '../../../../../shared/agent-session-pty-write-admission' import { assertLegacyAiVaultResumeCommandAllowed } from '../../../../ai-vault/structured-session-ownership' -import { InvalidArgumentError, defineMethod, type RpcAnyMethod } from '../../core' +import { InvalidArgumentError, defineMethod } from '../../core' import { isTerminalQueryReply } from '../../../../../shared/terminal-query-reply' import { assertTerminalAgentSendable } from '../../terminal-agent-send-guard' import { TerminalSend } from './unary-schemas' @@ -20,7 +20,7 @@ import { observeReplayedTerminalPrompt } from './terminal-prompt-receipt' -export const TERMINAL_SEND_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_SEND_METHODS = [ defineMethod({ name: 'terminal.send', params: TerminalSend, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts index bd16382d747..7572d41496f 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts @@ -1,4 +1,4 @@ -import { defineStreamingMethod, type RpcAnyMethod } from '../../core' +import { defineStreamingMethod } from '../../core' import { TerminalSubscribe } from './stream-schemas' import { isTerminalReadPayloadIncomplete } from './terminal-stream-replay' import { runTerminalBinarySubscription } from './terminal-legacy-subscribe-binary' @@ -8,7 +8,7 @@ import { } from './terminal-legacy-simple-subscriptions' import type { TerminalSubscriptionArgs } from './terminal-legacy-subscription-types' -export const TERMINAL_SUBSCRIBE_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_SUBSCRIBE_METHODS = [ // Streams live terminal output over WebSocket; mobile clients pass client+viewport for server-side auto-fit. defineStreamingMethod({ name: 'terminal.subscribe', diff --git a/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts index d71ac53e249..91bdf25d84d 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../../core' +import { defineMethod } from '../../core' import { TerminalHandle } from './unary-schemas' import { TerminalSetAutoRestoreFit, @@ -9,7 +9,7 @@ import { import { updateViewportForClient } from './terminal-viewport-update' import { TerminalGetAutoRestoreFitParams } from '../../../../../shared/rpc-contract/terminal-viewport-methods-params' -export const TERMINAL_VIEWPORT_METHODS_BEFORE_STREAMS: RpcAnyMethod[] = [ +export const TERMINAL_VIEWPORT_METHODS_BEFORE_STREAMS = [ defineMethod({ name: 'terminal.setDisplayMode', params: TerminalSetDisplayMode, @@ -78,7 +78,7 @@ export const TERMINAL_VIEWPORT_METHODS_BEFORE_STREAMS: RpcAnyMethod[] = [ }) ] -export const TERMINAL_VIEWPORT_METHODS_AFTER_STREAMS: RpcAnyMethod[] = [ +export const TERMINAL_VIEWPORT_METHODS_AFTER_STREAMS = [ defineMethod({ name: 'terminal.unsubscribe', params: TerminalUnsubscribe, diff --git a/src/main/runtime/rpc/methods/updater.test.ts b/src/main/runtime/rpc/methods/updater.test.ts index 9c3ce0ef810..1a925cbe767 100644 --- a/src/main/runtime/rpc/methods/updater.test.ts +++ b/src/main/runtime/rpc/methods/updater.test.ts @@ -1,4 +1,5 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' +import { eraseRpcMethods, type RpcMethodDeclaration } from '../core' import { configureRemoteServerUpdater } from '../../remote-server-updater' import { STATUS_METHODS } from './status' import { UPDATER_METHODS } from './updater' @@ -10,8 +11,8 @@ const snapshot = { status: { state: 'available', version: '1.5.1', changelog: null } } as const -function handler(methods: typeof UPDATER_METHODS, name: string) { - const method = methods.find((candidate) => candidate.name === name) +function handler(methods: readonly RpcMethodDeclaration[], name: string) { + const method = eraseRpcMethods(methods).find((candidate) => candidate.name === name) if (!method) { throw new Error(`Missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/updater.ts b/src/main/runtime/rpc/methods/updater.ts index 357f07a8c09..a14fb5ab6a4 100644 --- a/src/main/runtime/rpc/methods/updater.ts +++ b/src/main/runtime/rpc/methods/updater.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { checkRemoteServerUpdater, downloadRemoteServerUpdater, @@ -7,7 +7,7 @@ import { } from '../../remote-server-updater' import { UpdaterCheckParams } from '../../../../shared/rpc-contract/updater-params' -export const UPDATER_METHODS: RpcMethod[] = [ +export const UPDATER_METHODS = [ defineMethod({ name: 'updater.getStatus', params: null, diff --git a/src/main/runtime/rpc/methods/workspace-ports.ts b/src/main/runtime/rpc/methods/workspace-ports.ts index 5778f25732f..c96c91b436d 100644 --- a/src/main/runtime/rpc/methods/workspace-ports.ts +++ b/src/main/runtime/rpc/methods/workspace-ports.ts @@ -1,10 +1,10 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { WorkspacePortKillParams, WorkspacePortScanParams } from '../../../../shared/rpc-contract/workspace-ports-params' -export const WORKSPACE_PORT_METHODS: RpcMethod[] = [ +export const WORKSPACE_PORT_METHODS = [ defineMethod({ name: 'workspacePorts.scan', params: WorkspacePortScanParams, diff --git a/src/main/runtime/rpc/methods/worktree-catalog-methods.ts b/src/main/runtime/rpc/methods/worktree-catalog-methods.ts index 2010a219b7c..8b230c169d8 100644 --- a/src/main/runtime/rpc/methods/worktree-catalog-methods.ts +++ b/src/main/runtime/rpc/methods/worktree-catalog-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { resolveWorktreeCatalogSnapshot } from '../worktree-catalog-snapshot' import { supportsWorktreeVisibilitySourceDefaults } from '../worktree-visibility-client-capability' import { @@ -7,7 +7,7 @@ import { WorktreePsParams } from './worktree-schemas' -export const WORKTREE_CATALOG_METHODS: RpcMethod[] = [ +export const WORKTREE_CATALOG_METHODS = [ defineMethod({ name: 'worktree.ps', params: WorktreePsParams, diff --git a/src/main/runtime/rpc/methods/worktree.ts b/src/main/runtime/rpc/methods/worktree.ts index b3d816496c3..be8a0983036 100644 --- a/src/main/runtime/rpc/methods/worktree.ts +++ b/src/main/runtime/rpc/methods/worktree.ts @@ -5,7 +5,7 @@ import { } from '../../../automations/workspace-provenance' import { buildCliWorkspaceProvenance } from '../../../../shared/cli-workspace-provenance' import { displayNameUpdatePinsLabel } from '../../../../shared/worktree/display-name-provenance' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { buildManagedWorktreeCreateArgs } from './worktree-create-args' import { resolvePairedCallerHostId } from './paired-caller-host-id' import { resolveRuntimeNavigationTarget } from '../../../../shared/runtime-navigation' @@ -24,7 +24,7 @@ import { } from './worktree-schemas' import { WORKTREE_CATALOG_METHODS } from './worktree-catalog-methods' -export const WORKTREE_METHODS: RpcMethod[] = [ +export const WORKTREE_METHODS = [ ...WORKTREE_CATALOG_METHODS, defineMethod({ name: 'worktree.teardownMissingTerminals', diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts b/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts index 3cd3c1a54fb..4923dec1dec 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts @@ -1,5 +1,5 @@ import type { OrcaRuntimeService } from '../orca-runtime' -import type { RpcAnyMethod } from '../rpc/core' +import type { RpcAnyMethodDeclaration } from '../rpc/core' import type { DeviceRegistry } from '../device-registry' import type { E2EEKeypair } from '../e2ee-keypair' import type { MobileSocketTransportMetadata } from '../rpc/mobile-socket-wiring' @@ -56,7 +56,7 @@ export type OrcaRuntimeRpcServerOptions = { // Why: test-only override for the ownership reclaim cadence. metadataOwnershipPollMs?: number // Why: tests may inject inert protocol stages before production authorization registers them. - methods?: readonly RpcAnyMethod[] + methods?: readonly RpcAnyMethodDeclaration[] } export type PairingOfferUnavailableReason =