mirror of
https://github.com/stablyai/orca.git
synced 2026-09-30 00:03:15 +00:00
refactor(ai-vault-search): consolidate the review fixes into one owner per rule
Second pass over the session-search delta. Keeps the staged-write, streaming capture, consent-gate and SQL-side filtering designs; removes the layers that had accumulated around them. Store and schema (schema 10): row visibility is two SQL views instead of four hand-written predicates; publish nulls messages.batch_id and drops the batch row, so the messages view is `batch_id IS NULL` and a recycled batch id can no longer hide published rows; `published` column and its index removed; WAL checkpoint guard taken off every read path and sampled on the write loop only; maintenance class split into three plain functions; cwd_key indexed and the scope filter switched to a range seek with an EXPLAIN plan assertion. Service and capture: one write shape (streamingCapture flag and the dead array branch deleted); configure applies policy to the store once; one shutdown path (dispose removed) with an ownership guard so a late close cannot unregister a replacement service's sink; unverifiable sources are returned with a caveat instead of filtered like deletions; a transient write failure no longer pins the indexing badge at error; refresh lane reuses stableInFlightKey; message channel handles concurrent checkpoints. Query: relevance sort now groups by session before the candidate limit (the must-fix was only applied to newest); one operator parser shared by panel and backend with the apostrophe bug fixed; OR within a key, AND across keys; one documented case rule; received enums tolerate unknown values; result type derived from the schema; projection applied on the desktop IPC path; tokenizer contract pinned against fts5vocab. Remote and CLI: SSH and local ids rejected at the RPC boundary for all four methods; method-name regex sniffing removed from both transports; CommandSpec gained booleanFlags/repeatableFlags so args.ts carries no command vocabulary; settings update no longer blocks or fails on scanner reconfiguration; relay owner keeps its lease through caller cancellation and answers index-status on an unreadable policy; one SESSION_SEARCH_METHODS record feeds every caller. Renderer: coverage polling stops once the index settles and re-arms on focus; search results feed the coverage store instead of re-fetching; the `updating` state and header line are deleted in favour of the list's own loading state; local-only notice only when a remote host is in scope; status-bar segment is read-only and opens settings; ownerKey folded into the args key. Providers: OpenCode capture resets (not clears) the parse deadline; capture and preview reads degrade to a scan issue instead of dropping the session; a consumer write failure is not counted as a worker death; response union discriminated on `kind`; worker host split out of the client. Every behavioural fix carries a test that fails when the fix is reverted.
This commit is contained in:
@@ -4,19 +4,19 @@ import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import SyncDatabase from '../../src/main/sqlite/sync-database'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import { SessionSearchQuery } from '../../src/main/ai-vault-search/session-search-query'
|
||||
import { openSessionSearchDatabase } from '../../src/main/ai-vault-search/session-search-schema'
|
||||
import { deleteExpiredSearchFiles } from '../../src/main/ai-vault-search/session-search-retention-delete'
|
||||
import { SearchWalBackpressureError } from '../../src/main/ai-vault-search/session-search-wal-budget'
|
||||
|
||||
// Bundle with esbuild --bundle --platform=node, then run on the host under test.
|
||||
const root = await mkdtemp(join(tmpdir(), 'orca-search-retention-bench-'))
|
||||
try {
|
||||
for (const mode of ['whole-file', 'batched', 'batched-pinned-reader']) {
|
||||
const path = join(root, `${mode}.sqlite`)
|
||||
const store = new SessionSearchStore(path)
|
||||
const db = openSessionSearchDatabase(path)
|
||||
let reader: SyncDatabase | null = null
|
||||
try {
|
||||
const db = store.db
|
||||
const query = new SessionSearchQuery(db)
|
||||
db.exec(`INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,resume_command)
|
||||
VALUES (1,'claude','1','fixture','synthetic benchmark','/fixture','/fixture','');
|
||||
INSERT INTO files(path,byte_offset,mtime_ms,session_row_id) VALUES ('fixture',1,1,1);
|
||||
@@ -28,7 +28,7 @@ try {
|
||||
FROM messages;
|
||||
INSERT INTO conversation_fts(rowid,user_text) SELECT rowid,user_text FROM messages_fts;
|
||||
COMMIT; PRAGMA wal_checkpoint(TRUNCATE)`)
|
||||
assert.equal(store.search({ query: 'needle' }).hits.length, 1)
|
||||
assert.equal(query.execute({ query: 'needle' }, null).hits.length, 1)
|
||||
if (mode === 'batched-pinned-reader') {
|
||||
reader = new SyncDatabase(path, { readonly: true })
|
||||
reader.exec('BEGIN')
|
||||
@@ -56,29 +56,11 @@ try {
|
||||
() => {},
|
||||
async () => {
|
||||
intervals.push(performance.now() - previous)
|
||||
assert.equal(store.search({ query: 'needle' }).hits.length, 0)
|
||||
assert.equal(query.execute({ query: 'needle' }, null).hits.length, 0)
|
||||
await yieldToEventLoop()
|
||||
previous = performance.now()
|
||||
}
|
||||
).catch(async (error: unknown) => {
|
||||
if (!(error instanceof SearchWalBackpressureError) || !reader) {
|
||||
throw error
|
||||
}
|
||||
console.log(
|
||||
JSON.stringify({
|
||||
mode,
|
||||
backpressured: true,
|
||||
walBytes: (await stat(`${path}-wal`)).size
|
||||
})
|
||||
)
|
||||
reader.exec('COMMIT')
|
||||
await deleteExpiredSearchFiles(
|
||||
db,
|
||||
null,
|
||||
() => false,
|
||||
() => {}
|
||||
)
|
||||
})
|
||||
)
|
||||
}
|
||||
const wallMs = performance.now() - started
|
||||
assert.equal(
|
||||
@@ -106,7 +88,7 @@ try {
|
||||
)
|
||||
} finally {
|
||||
reader?.close()
|
||||
store.close()
|
||||
db.close()
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
|
||||
@@ -3,16 +3,19 @@ import { mkdtemp, rm, stat } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store'
|
||||
import { deleteExpiredSearchFiles } from '../../src/main/ai-vault-search/session-search-retention-delete'
|
||||
import { SessionSearchIndexWriter } from '../../src/main/ai-vault-search/session-search-index-writer'
|
||||
import { stagedWriteUpdate } from '../../src/main/ai-vault-search/session-search-staged-write-fixtures'
|
||||
import { SessionSearchQuery } from '../../src/main/ai-vault-search/session-search-query'
|
||||
import { openSessionSearchDatabase } from '../../src/main/ai-vault-search/session-search-schema'
|
||||
import { stagedWriteUpdate } from '../../src/main/ai-vault-search/session-search-staged-write-test-fixture'
|
||||
|
||||
const root = await mkdtemp(join(tmpdir(), 'orca-search-write-bench-'))
|
||||
try {
|
||||
const path = join(root, 'index.sqlite')
|
||||
const store = new SessionSearchStore(path)
|
||||
const db = openSessionSearchDatabase(path)
|
||||
try {
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const writer = new SessionSearchIndexWriter(db)
|
||||
const query = new SessionSearchQuery(db)
|
||||
for (const mode of ['replace', 'append', 'replace'] as const) {
|
||||
const update = stagedWriteUpdate(
|
||||
`benchmarkneedle ${'synthetic coding context src/example.ts '.repeat(5)}`,
|
||||
@@ -22,20 +25,18 @@ try {
|
||||
const steps: number[] = []
|
||||
let before = performance.now()
|
||||
const start = before
|
||||
await writer.apply(
|
||||
update,
|
||||
() => true,
|
||||
async () => {
|
||||
await writer.apply(update, {
|
||||
yieldStep: async () => {
|
||||
steps.push(performance.now() - before)
|
||||
await yieldToEventLoop()
|
||||
before = performance.now()
|
||||
}
|
||||
)
|
||||
})
|
||||
const wallMs = performance.now() - start
|
||||
if (!steps.length) {
|
||||
steps.push(wallMs)
|
||||
}
|
||||
assert.equal(store.search({ query: 'benchmarkneedle' }).hits.length, 1)
|
||||
assert.equal(query.execute({ query: 'benchmarkneedle' }, null).hits.length, 1)
|
||||
console.log(
|
||||
JSON.stringify({
|
||||
platform: process.platform,
|
||||
@@ -48,10 +49,15 @@ try {
|
||||
walBytes: (await stat(`${path}-wal`)).size
|
||||
})
|
||||
)
|
||||
await store.purgeOlderThan(null)
|
||||
await deleteExpiredSearchFiles(
|
||||
db,
|
||||
null,
|
||||
() => false,
|
||||
() => {}
|
||||
)
|
||||
}
|
||||
} finally {
|
||||
store.close()
|
||||
db.close()
|
||||
}
|
||||
} finally {
|
||||
await rm(root, { recursive: true, force: true })
|
||||
|
||||
@@ -5,6 +5,7 @@ import {
|
||||
import type { AiVaultSearchIndexStatus } from '../shared/ai-vault-search-settings'
|
||||
import { aiVaultAgentLabel } from '../shared/ai-vault-types'
|
||||
import { aiVaultSearchUnindexedProviders } from '../shared/ai-vault-search-coverage'
|
||||
import { getRuntimePathBasename } from '../shared/cross-platform-path'
|
||||
import type { AiVaultSearchHit, AiVaultSearchResult } from '../shared/ai-vault-search-types'
|
||||
|
||||
const ROLE_LABEL: Record<AiVaultSearchHit['evidence']['role'], string> = {
|
||||
@@ -40,7 +41,9 @@ function relativeAge(iso: string | null, now = Date.now()): string {
|
||||
}
|
||||
|
||||
function projectLabel(hit: AiVaultSearchHit): string {
|
||||
const cwd = hit.cwd ? (hit.cwd.replaceAll('\\', '/').split('/').findLast(Boolean) ?? '—') : '—'
|
||||
// Why: the path comes from the execution host, so node:path would read a
|
||||
// Windows path with POSIX rules (and the reverse) when the two disagree.
|
||||
const cwd = (hit.cwd ? getRuntimePathBasename(hit.cwd) : '') || '—'
|
||||
return hit.branch ? `${cwd} · ${hit.branch}` : cwd
|
||||
}
|
||||
|
||||
|
||||
+49
-17
@@ -5,7 +5,8 @@ import {
|
||||
CLI_BOOLEAN_FLAGS,
|
||||
CLI_GLOBAL_FLAGS,
|
||||
CLI_GLOBAL_VALUE_FLAGS,
|
||||
findCliCommandIndex
|
||||
findCliCommandIndex,
|
||||
findCliCommandPathAt
|
||||
} from '../shared/cli-argument-boundary'
|
||||
|
||||
export { specPaths }
|
||||
@@ -28,23 +29,60 @@ function setFlagValue(
|
||||
flags: Map<string, string | boolean>,
|
||||
name: string,
|
||||
value: string,
|
||||
search = false
|
||||
repeatable: ReadonlySet<string>
|
||||
): void {
|
||||
const existing = flags.get(name)
|
||||
if (
|
||||
typeof existing === 'string' &&
|
||||
(REPEATABLE_STRING_FLAGS.has(name) || (search && (name === 'agent' || name === 'path')))
|
||||
) {
|
||||
if (typeof existing === 'string' && repeatable.has(name)) {
|
||||
flags.set(name, `${existing}${REPEATED_FLAG_SEPARATOR}${value}`)
|
||||
return
|
||||
}
|
||||
flags.set(name, value)
|
||||
}
|
||||
|
||||
export function parseArgs(argv: string[], commandPaths?: readonly string[][]): ParsedArgs {
|
||||
/** The most specific spec whose path prefixes `path`, so a group never shadows a leaf. */
|
||||
function specForPathPrefix(
|
||||
specs: readonly CommandSpec[],
|
||||
path: readonly string[]
|
||||
): CommandSpec | undefined {
|
||||
let best: { spec: CommandSpec; length: number } | undefined
|
||||
for (const spec of specs) {
|
||||
for (const candidate of specPaths(spec)) {
|
||||
if (
|
||||
candidate.length <= path.length &&
|
||||
candidate.every((part, index) => part === path[index]) &&
|
||||
(!best || candidate.length > best.length)
|
||||
) {
|
||||
best = { spec, length: candidate.length }
|
||||
}
|
||||
}
|
||||
}
|
||||
return best?.spec
|
||||
}
|
||||
|
||||
export function parseArgs(
|
||||
argv: string[],
|
||||
commandPaths?: readonly string[][],
|
||||
specs: readonly CommandSpec[] = []
|
||||
): ParsedArgs {
|
||||
const commandPath: string[] = []
|
||||
const flags = new Map<string, string | boolean>()
|
||||
const commandIndex = findCliCommandIndex(argv, commandPaths ?? [])
|
||||
const paths = commandPaths ?? []
|
||||
// Why: the boundary scan and the flag reader must agree on which flags take no
|
||||
// value, so both read the global set widened by every spec's own vocabulary.
|
||||
const allBooleanFlags = new Set([
|
||||
...BOOLEAN_FLAGS,
|
||||
...specs.flatMap((spec) => spec.booleanFlags ?? [])
|
||||
])
|
||||
const commandIndex = findCliCommandIndex(argv, paths, [], allBooleanFlags)
|
||||
const pinned =
|
||||
commandIndex === -1 ? null : findCliCommandPathAt(argv, paths, commandIndex, allBooleanFlags)
|
||||
// Resolved lazily: without a registry the command is only known once its
|
||||
// leading tokens have been read.
|
||||
const activeSpec = (): CommandSpec | undefined => specForPathPrefix(specs, pinned ?? commandPath)
|
||||
const repeatableFlags = (): ReadonlySet<string> => {
|
||||
const scoped = activeSpec()?.repeatableFlags
|
||||
return scoped ? new Set([...REPEATABLE_STRING_FLAGS, ...scoped]) : REPEATABLE_STRING_FLAGS
|
||||
}
|
||||
|
||||
for (let i = 0; i < argv.length; i += 1) {
|
||||
const token = argv[i]
|
||||
@@ -63,19 +101,13 @@ export function parseArgs(argv: string[], commandPaths?: readonly string[][]): P
|
||||
flags,
|
||||
assignment.slice(0, equalsIndex),
|
||||
assignment.slice(equalsIndex + 1),
|
||||
(argv[commandIndex] ?? commandPath[0]) === 'search'
|
||||
repeatableFlags()
|
||||
)
|
||||
continue
|
||||
}
|
||||
|
||||
const flag = assignment
|
||||
if (
|
||||
BOOLEAN_FLAGS.has(flag) ||
|
||||
((argv[commandIndex] ?? commandPath[0]) === 'search' &&
|
||||
['enable', 'disable', 'clear-index', 'index-status', 'pause', 'resume-indexing'].includes(
|
||||
flag
|
||||
))
|
||||
) {
|
||||
if (BOOLEAN_FLAGS.has(flag) || (activeSpec()?.booleanFlags?.includes(flag) ?? false)) {
|
||||
flags.set(flag, true)
|
||||
continue
|
||||
}
|
||||
@@ -90,7 +122,7 @@ export function parseArgs(argv: string[], commandPaths?: readonly string[][]): P
|
||||
flags.set(flag, true)
|
||||
continue
|
||||
}
|
||||
setFlagValue(flags, flag, next, (argv[commandIndex] ?? commandPath[0]) === 'search')
|
||||
setFlagValue(flags, flag, next, repeatableFlags())
|
||||
i += 1
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
import { expect, it } from 'vitest'
|
||||
import { parseArgs, REPEATED_FLAG_SEPARATOR, type CommandSpec } from './args'
|
||||
import { COMMAND_SPECS } from './specs'
|
||||
|
||||
// A command other than `search` on purpose: the parser must read this vocabulary
|
||||
// off the resolved spec, not off a command name written into the parser.
|
||||
const DEMO: CommandSpec = {
|
||||
path: ['demo', 'run'],
|
||||
summary: 'demo',
|
||||
usage: 'demo run',
|
||||
allowedFlags: ['enable', 'agent', 'note'],
|
||||
booleanFlags: ['enable'],
|
||||
repeatableFlags: ['agent']
|
||||
}
|
||||
|
||||
it('reads value-less and repeatable flags from the spec that owns them', () => {
|
||||
const parsed = parseArgs(
|
||||
['demo', 'run', '--enable', '--agent', 'codex', '--agent', 'claude', '--note', 'hi'],
|
||||
[DEMO.path],
|
||||
[DEMO]
|
||||
)
|
||||
|
||||
expect(parsed.commandPath).toEqual(['demo', 'run'])
|
||||
expect(parsed.flags.get('enable')).toBe(true)
|
||||
expect(parsed.flags.get('agent')).toBe(`codex${REPEATED_FLAG_SEPARATOR}claude`)
|
||||
expect(parsed.flags.get('note')).toBe('hi')
|
||||
})
|
||||
|
||||
it('finds the command path behind a spec-declared boolean flag', () => {
|
||||
const parsed = parseArgs(['--enable', 'demo', 'run'], [DEMO.path], [DEMO])
|
||||
|
||||
expect(parsed.commandPath).toEqual(['demo', 'run'])
|
||||
expect(parsed.flags.get('enable')).toBe(true)
|
||||
})
|
||||
|
||||
it('does not leak one command vocabulary into another', () => {
|
||||
const parsed = parseArgs(
|
||||
['worktree', 'create', '--agent', 'codex', '--agent', 'claude'],
|
||||
COMMAND_SPECS.flatMap((spec) => [spec.path]),
|
||||
COMMAND_SPECS
|
||||
)
|
||||
|
||||
expect(parsed.flags.get('agent')).toBe('claude')
|
||||
})
|
||||
@@ -9,6 +9,11 @@ export type CommandSpec = {
|
||||
summary: string
|
||||
usage: string
|
||||
allowedFlags: string[]
|
||||
// Why: value-less and repeatable flags are per-command vocabulary. Declaring
|
||||
// them here keeps one command's flags out of the global parser, which cannot
|
||||
// scope `--agent` (repeatable for `search`, single-valued for `worktree create`).
|
||||
booleanFlags?: string[]
|
||||
repeatableFlags?: string[]
|
||||
positionalArgs?: string[]
|
||||
examples?: string[]
|
||||
notes?: string[]
|
||||
|
||||
+10
-14
@@ -14,11 +14,11 @@ import {
|
||||
import { listSshTargets, findSshTargetByName } from '../host-selector-alternatives'
|
||||
import { searchAllHosts } from '../session-search-all-hosts'
|
||||
import {
|
||||
searchHostMethod,
|
||||
createSearchHostCall,
|
||||
SEARCH_ALL_TIMEOUT_MS,
|
||||
type SearchHost
|
||||
} from '../session-search-host-query'
|
||||
import { waitForPromiseWithSignal } from '../../shared/abort-signal-reason'
|
||||
import type { SessionSearchOperation } from '../../shared/ai-vault-search-rpc-methods'
|
||||
|
||||
export const SEARCH_DISABLED_MESSAGE =
|
||||
'Session search is off. Enable it in Settings > Agent Session History, or run `orca search --agent-session --enable`.'
|
||||
@@ -39,10 +39,6 @@ export const SEARCH_HANDLERS: Record<string, CommandHandler> = {
|
||||
controller.abort(new Error('Search interrupted.'))
|
||||
}
|
||||
process.once('SIGINT', interrupt)
|
||||
const options = (): { signal: AbortSignal; timeoutMs: number } => ({
|
||||
signal: controller.signal,
|
||||
timeoutMs: Math.max(1, deadline - Date.now())
|
||||
})
|
||||
try {
|
||||
if (command.host === 'all') {
|
||||
const result = await searchAllHosts(client, command, controller.signal, deadline)
|
||||
@@ -96,16 +92,16 @@ export const SEARCH_HANDLERS: Record<string, CommandHandler> = {
|
||||
host.targetId = target.id
|
||||
host.name = target.label
|
||||
}
|
||||
const target = host.targetId
|
||||
? { targetId: host.targetId }
|
||||
: command.host?.kind === 'runtime'
|
||||
// Why: `--host runtime:<env>` already selected that runtime's transport, so
|
||||
// the id only restamps the answer; the targetId spread lives in the shared
|
||||
// call factory with the method routing and the deadline.
|
||||
const stamp =
|
||||
!host.targetId && command.host?.kind === 'runtime'
|
||||
? { executionHostId: command.host.id }
|
||||
: {}
|
||||
const call = (operation: 'query' | 'status' | 'configure', params: object) =>
|
||||
waitForPromiseWithSignal(
|
||||
client.call(searchHostMethod(host, operation), { ...params, ...target }, options()),
|
||||
controller.signal
|
||||
)
|
||||
const send = createSearchHostCall(host, controller.signal, deadline)
|
||||
const call = (operation: SessionSearchOperation, params: object) =>
|
||||
send(operation, { ...params, ...stamp })
|
||||
if (command.configure || command.status) {
|
||||
const response = await call(
|
||||
command.configure ? 'configure' : 'status',
|
||||
|
||||
+4
-1
@@ -83,7 +83,10 @@ export async function main(
|
||||
await runClaudeTeams(argv.slice(1), cwd)
|
||||
return
|
||||
}
|
||||
const parsed = normalizeCommandPositionals(COMMAND_SPECS, parseArgs(argv, COMMAND_PATHS))
|
||||
const parsed = normalizeCommandPositionals(
|
||||
COMMAND_SPECS,
|
||||
parseArgs(argv, COMMAND_PATHS, COMMAND_SPECS)
|
||||
)
|
||||
const helpPath = resolveHelpPath(parsed)
|
||||
if (helpPath !== null) {
|
||||
printHelp(COMMAND_SPECS, helpPath)
|
||||
|
||||
@@ -92,36 +92,6 @@ describe.skipIf(process.platform === 'win32')('runtime transport', () => {
|
||||
}
|
||||
})
|
||||
|
||||
it('rejects an oversized search frame before buffering the complete response', async () => {
|
||||
const directory = mkdtempSync(join(tmpdir(), 'orca-search-size-'))
|
||||
const endpoint = join(directory, 'runtime.sock')
|
||||
const server = createServer((socket) => {
|
||||
sockets.add(socket)
|
||||
socket.on('error', () => undefined)
|
||||
socket.once('close', () => sockets.delete(socket))
|
||||
socket.once('data', () => socket.write(Buffer.alloc(4 * 1024 * 1024 + 1, 'a')))
|
||||
})
|
||||
servers.add(server)
|
||||
await new Promise<void>((resolve) => server.listen(endpoint, resolve))
|
||||
try {
|
||||
await expect(
|
||||
sendRequest(
|
||||
{
|
||||
runtimeId: 'test',
|
||||
pid: 1,
|
||||
transports: [{ kind: 'unix', endpoint }],
|
||||
authToken: 'fixture',
|
||||
startedAt: 1
|
||||
},
|
||||
'aiVault.searchSessions',
|
||||
{ query: 'fixture' },
|
||||
30_000
|
||||
)
|
||||
).rejects.toMatchObject({ code: 'invalid_runtime_response' })
|
||||
} finally {
|
||||
rmSync(directory, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
it('refreshes the per-call timeout when the runtime sends keepalive frames', async () => {
|
||||
const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-transport-'))
|
||||
const endpoint = join(userDataPath, 'runtime.sock')
|
||||
@@ -211,3 +181,46 @@ describe.skipIf(process.platform === 'win32')('runtime transport', () => {
|
||||
expect(Date.now() - start).toBeLessThan(5000)
|
||||
})
|
||||
})
|
||||
|
||||
it('reads a large response frame on any method, without a per-method size rule', async () => {
|
||||
const directory = mkdtempSync(join(tmpdir(), 'orca-search-size-'))
|
||||
const endpoint = join(directory, 'runtime.sock')
|
||||
const padding = 'a'.repeat(5 * 1024 * 1024)
|
||||
const server = createServer((socket) => {
|
||||
sockets.add(socket)
|
||||
socket.on('error', () => undefined)
|
||||
socket.once('close', () => sockets.delete(socket))
|
||||
let request = ''
|
||||
socket.on('data', (chunk: Buffer) => {
|
||||
request += chunk.toString('utf8')
|
||||
const newline = request.indexOf('\n')
|
||||
if (newline === -1) {
|
||||
return
|
||||
}
|
||||
const { id } = JSON.parse(request.slice(0, newline))
|
||||
socket.write(
|
||||
`${JSON.stringify({ id, ok: true, result: { padding }, _meta: { runtimeId: 'test' } })}\n`
|
||||
)
|
||||
})
|
||||
})
|
||||
servers.add(server)
|
||||
await new Promise<void>((resolve) => server.listen(endpoint, resolve))
|
||||
try {
|
||||
const response = await sendRequest<{ padding: string }>(
|
||||
{
|
||||
runtimeId: 'test',
|
||||
pid: 1,
|
||||
transports: [{ kind: 'unix', endpoint }],
|
||||
authToken: 'fixture',
|
||||
startedAt: 1
|
||||
},
|
||||
'aiVault.searchSessions',
|
||||
{ query: 'fixture' },
|
||||
30_000
|
||||
)
|
||||
expect(response.ok).toBe(true)
|
||||
expect(response.ok === true && response.result.padding.length).toBe(padding.length)
|
||||
} finally {
|
||||
rmSync(directory, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
@@ -35,9 +35,6 @@ export async function sendRequest<TResult>(
|
||||
}
|
||||
const socket = createConnection(transport.endpoint)
|
||||
let lineSegments: string[] = []
|
||||
let lineBytes = 0
|
||||
const searchResponseLimit =
|
||||
method.startsWith('aiVault.') && /search/i.test(method) ? 4 * 1024 * 1024 : Infinity
|
||||
let settled = false
|
||||
const requestId = randomUUID()
|
||||
|
||||
@@ -80,10 +77,6 @@ export async function sendRequest<TResult>(
|
||||
socket.destroy()
|
||||
}
|
||||
signal?.addEventListener('abort', onAbort, { once: true })
|
||||
if (signal?.aborted) {
|
||||
onAbort()
|
||||
return
|
||||
}
|
||||
socket.setEncoding('utf8')
|
||||
socket.once('error', () => {
|
||||
finish({
|
||||
@@ -116,20 +109,6 @@ export async function sendRequest<TResult>(
|
||||
let cursor = 0
|
||||
while (cursor < chunk.length && !settled) {
|
||||
const newlineIndex = chunk.indexOf('\n', cursor)
|
||||
lineBytes += Buffer.byteLength(
|
||||
chunk.slice(cursor, newlineIndex === -1 ? undefined : newlineIndex)
|
||||
)
|
||||
if (lineBytes > searchResponseLimit) {
|
||||
finish({
|
||||
ok: false,
|
||||
error: new RuntimeClientError(
|
||||
'invalid_runtime_response',
|
||||
'Search response exceeds the size limit.'
|
||||
)
|
||||
})
|
||||
socket.destroy()
|
||||
return
|
||||
}
|
||||
if (newlineIndex === -1) {
|
||||
lineSegments.push(chunk.slice(cursor))
|
||||
return
|
||||
@@ -142,7 +121,6 @@ export async function sendRequest<TResult>(
|
||||
lineSegments = []
|
||||
}
|
||||
cursor = newlineIndex + 1
|
||||
lineBytes = 0
|
||||
if (line.trim().length === 0) {
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -1,9 +1,15 @@
|
||||
import { expect, it } from 'vitest'
|
||||
import { parseArgs, REPEATED_FLAG_SEPARATOR } from './args'
|
||||
import { parseSearchCommand } from './search-command-arguments'
|
||||
import { SEARCH_COMMAND_SPECS } from './specs/search'
|
||||
|
||||
// The search flag vocabulary lives on its spec, so the parser only knows the
|
||||
// repeatable and value-less flags when the registry is handed to it.
|
||||
const parseSearchArgs = (argv: string[]): ReturnType<typeof parseArgs> =>
|
||||
parseArgs(argv, [['search']], SEARCH_COMMAND_SPECS)
|
||||
|
||||
it('preserves repeated filters through argv and validates before configuration', () => {
|
||||
const parsed = parseArgs([
|
||||
const parsed = parseSearchArgs([
|
||||
'search',
|
||||
'--agent-session',
|
||||
'needle',
|
||||
@@ -34,33 +40,32 @@ it('preserves repeated filters through argv and validates before configuration',
|
||||
|
||||
it('accepts queryless policy management and refuses aggregate mutations', () => {
|
||||
expect(
|
||||
parseSearchCommand(parseArgs(['search', '--agent-session', '--enable']).flags).configure
|
||||
parseSearchCommand(parseSearchArgs(['search', '--agent-session', '--enable']).flags).configure
|
||||
).toEqual({ enabled: true })
|
||||
expect(
|
||||
parseSearchCommand(
|
||||
parseArgs(['search', '--disable', '--clear-index', '--host', 'ssh:box']).flags
|
||||
parseSearchArgs(['search', '--disable', '--clear-index', '--host', 'ssh:box']).flags
|
||||
).configure
|
||||
).toEqual({ enabled: false, clearIndex: true })
|
||||
expect(() =>
|
||||
parseSearchCommand(parseArgs(['search', '--enable', '--host', 'all']).flags)
|
||||
parseSearchCommand(parseSearchArgs(['search', '--enable', '--host', 'all']).flags)
|
||||
).toThrow()
|
||||
expect(() =>
|
||||
parseSearchCommand(parseSearchArgs(['search', '--enable', '--disable']).flags)
|
||||
).toThrow()
|
||||
expect(() => parseSearchCommand(parseArgs(['search', '--enable', '--disable']).flags)).toThrow()
|
||||
})
|
||||
|
||||
it('handles command discovery, equals syntax, Windows paths and host-specific scope rules', () => {
|
||||
const parsed = parseArgs(
|
||||
[
|
||||
'--json',
|
||||
'search',
|
||||
'--agent-session=needle',
|
||||
'--agent=codex',
|
||||
'--agent=claude',
|
||||
'--path=C:\\work',
|
||||
'--path=\\\\server\\share',
|
||||
'--host=all'
|
||||
],
|
||||
[['search']]
|
||||
)
|
||||
const parsed = parseSearchArgs([
|
||||
'--json',
|
||||
'search',
|
||||
'--agent-session=needle',
|
||||
'--agent=codex',
|
||||
'--agent=claude',
|
||||
'--path=C:\\work',
|
||||
'--path=\\\\server\\share',
|
||||
'--host=all'
|
||||
])
|
||||
expect(parseSearchCommand(parsed.flags).query).toMatchObject({
|
||||
agents: ['codex', 'claude'],
|
||||
scopePaths: ['C:\\work', '\\\\server\\share']
|
||||
@@ -72,6 +77,6 @@ it('handles command discovery, equals syntax, Windows paths and host-specific sc
|
||||
['--agent-session=needle', '--host=all', '--path=~/private'],
|
||||
['--index-status', '--agent-session=needle']
|
||||
]) {
|
||||
expect(() => parseSearchCommand(parseArgs(['search', ...args], [['search']]).flags)).toThrow()
|
||||
expect(() => parseSearchCommand(parseSearchArgs(['search', ...args]).flags)).toThrow()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -137,29 +137,25 @@ export async function searchAllHosts(
|
||||
message: 'Overall search deadline exceeded.'
|
||||
}
|
||||
}
|
||||
const controller = new AbortController()
|
||||
const timer = setTimeout(
|
||||
() => controller.abort(new Error('Search host deadline exceeded.')),
|
||||
Math.min(SEARCH_HOST_TIMEOUT_MS, deadline - Date.now())
|
||||
// Why: querySearchHost owns the per-host deadline; arming a second one here
|
||||
// produced two timers and two error messages for one call.
|
||||
const result = await querySearchHost(
|
||||
host,
|
||||
command,
|
||||
true,
|
||||
signal,
|
||||
Math.min(deadline, Date.now() + SEARCH_HOST_TIMEOUT_MS)
|
||||
)
|
||||
const abort = (): void => controller.abort(signal.reason)
|
||||
signal.addEventListener('abort', abort, { once: true })
|
||||
try {
|
||||
const result = await querySearchHost(host, command, true, controller.signal)
|
||||
const bytes = Buffer.byteLength(JSON.stringify(result))
|
||||
if (bytes > bytesRemaining) {
|
||||
return {
|
||||
host: result.host,
|
||||
outcome: 'omitted',
|
||||
message: 'Aggregate response limit reached.'
|
||||
}
|
||||
const bytes = Buffer.byteLength(JSON.stringify(result))
|
||||
if (bytes > bytesRemaining) {
|
||||
return {
|
||||
host: result.host,
|
||||
outcome: 'omitted',
|
||||
message: 'Aggregate response limit reached.'
|
||||
}
|
||||
bytesRemaining -= bytes
|
||||
return result
|
||||
} finally {
|
||||
clearTimeout(timer)
|
||||
signal.removeEventListener('abort', abort)
|
||||
}
|
||||
bytesRemaining -= bytes
|
||||
return result
|
||||
})
|
||||
const combined = [...results, ...skipped]
|
||||
if (
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
import type { RuntimeClient } from './runtime-client'
|
||||
import type { RuntimeRpcSuccess } from './runtime/types'
|
||||
import {
|
||||
SessionSearchResultSchema,
|
||||
SessionSearchStatusSchema
|
||||
} from '../shared/ai-vault-search-contract'
|
||||
import type { AiVaultSearchResult } from '../shared/ai-vault-search-types'
|
||||
import {
|
||||
SESSION_SEARCH_METHODS,
|
||||
type SessionSearchOperation
|
||||
} from '../shared/ai-vault-search-rpc-methods'
|
||||
import type { SearchCommand } from './search-command-arguments'
|
||||
import { waitForPromiseWithSignal } from '../shared/abort-signal-reason'
|
||||
|
||||
@@ -33,46 +38,53 @@ export const SEARCH_ALL_HOST_LIMIT = 16
|
||||
|
||||
export function searchHostMethod(
|
||||
host: Pick<SearchHost, 'targetId'>,
|
||||
operation: 'query' | 'status' | 'configure'
|
||||
operation: SessionSearchOperation
|
||||
): string {
|
||||
if (host.targetId) {
|
||||
return `aiVault.sshSearch${{ query: 'Sessions', status: 'IndexStatus', configure: 'Configure' }[operation]}`
|
||||
const methods = SESSION_SEARCH_METHODS[operation]
|
||||
return host.targetId ? methods.runtimeSsh : methods.runtime
|
||||
}
|
||||
|
||||
/**
|
||||
* The one place that knows a search host's method routing, its target spread and
|
||||
* its deadline, so the single-host and all-hosts paths cannot arm two of any of
|
||||
* them for the same call.
|
||||
*/
|
||||
export function createSearchHostCall(
|
||||
host: Pick<SearchHost, 'client' | 'targetId'>,
|
||||
signal: AbortSignal,
|
||||
deadline: number
|
||||
): (operation: SessionSearchOperation, params?: object) => Promise<RuntimeRpcSuccess<unknown>> {
|
||||
return (operation, params = {}) => {
|
||||
const remaining = deadline - Date.now()
|
||||
if (remaining <= 0) {
|
||||
return Promise.reject(new Error('Search host deadline exceeded.'))
|
||||
}
|
||||
return waitForPromiseWithSignal(
|
||||
host.client.call(
|
||||
searchHostMethod(host, operation),
|
||||
{ ...params, ...(host.targetId ? { targetId: host.targetId } : {}) },
|
||||
{ timeoutMs: remaining, signal }
|
||||
),
|
||||
signal
|
||||
)
|
||||
}
|
||||
return {
|
||||
query: 'aiVault.searchSessions',
|
||||
status: 'aiVault.searchIndexStatus',
|
||||
configure: 'aiVault.configureSessionSearch'
|
||||
}[operation]
|
||||
}
|
||||
|
||||
export async function querySearchHost(
|
||||
host: SearchHost,
|
||||
command: SearchCommand,
|
||||
aggregate: boolean,
|
||||
signal: AbortSignal
|
||||
signal: AbortSignal,
|
||||
deadline = Date.now() + SEARCH_HOST_TIMEOUT_MS
|
||||
): Promise<SearchHostResult> {
|
||||
const identity: SearchHostResult['host'] = {
|
||||
id: host.id,
|
||||
name: host.name,
|
||||
selector: host.selector
|
||||
}
|
||||
const deadline = Date.now() + SEARCH_HOST_TIMEOUT_MS
|
||||
const call = async (operation: 'query' | 'status', args: object = {}): Promise<unknown> => {
|
||||
const remaining = deadline - Date.now()
|
||||
if (remaining <= 0) {
|
||||
throw new Error('Search host deadline exceeded.')
|
||||
}
|
||||
const response = await waitForPromiseWithSignal(
|
||||
host.client.call(
|
||||
searchHostMethod(host, operation),
|
||||
{
|
||||
...args,
|
||||
...(host.targetId ? { targetId: host.targetId } : {})
|
||||
},
|
||||
{ timeoutMs: remaining, signal }
|
||||
),
|
||||
signal
|
||||
)
|
||||
const send = createSearchHostCall(host, signal, deadline)
|
||||
const call = async (operation: SessionSearchOperation, args: object = {}): Promise<unknown> => {
|
||||
const response = await send(operation, args)
|
||||
if (response._meta?.runtimeId) {
|
||||
identity.runtimeId = response._meta.runtimeId.slice(0, 512)
|
||||
}
|
||||
|
||||
@@ -23,6 +23,16 @@ export const SEARCH_COMMAND_SPECS: CommandSpec[] = [
|
||||
'pause',
|
||||
'resume-indexing'
|
||||
],
|
||||
booleanFlags: [
|
||||
'enable',
|
||||
'disable',
|
||||
'clear-index',
|
||||
'index-status',
|
||||
'pause',
|
||||
'resume-indexing',
|
||||
'newest'
|
||||
],
|
||||
repeatableFlags: ['agent', 'path'],
|
||||
notes: [
|
||||
'Searches what you typed, what the agent said, the commands it ran, and the first 3 KB of each tool output across Claude Code, Codex, Cursor, Gemini, OpenCode, and the other agents Orca scans.',
|
||||
'Quote paths, identifiers, or error text to match them exactly; plain words match anywhere. A misspelled word is repaired from the index vocabulary when nothing matches.',
|
||||
|
||||
@@ -5,7 +5,7 @@ import { tmpdir } from 'node:os'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { parseSearchCandidates } from './session-search-parse-candidates'
|
||||
import { sessionCandidate } from './session-search-transcript-fixtures'
|
||||
import { stagedWriteUpdate } from './session-search-staged-write-fixtures'
|
||||
import { stagedWriteUpdate } from './session-search-staged-write-test-fixture'
|
||||
import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture'
|
||||
import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache'
|
||||
import * as sourceRead from '../native-chat/wsl-transcript-fs-access'
|
||||
@@ -38,7 +38,9 @@ it('preserves published content and cursor when a whole-JSON refresh is canceled
|
||||
return text
|
||||
})
|
||||
const controller = new AbortController()
|
||||
parsing = parseSearchCandidates(store, [candidate], controller.signal).catch((error) => error)
|
||||
parsing = parseSearchCandidates(store, [candidate], { signal: controller.signal }).catch(
|
||||
(error) => error
|
||||
)
|
||||
await reached.promise
|
||||
controller.abort()
|
||||
held.resolve()
|
||||
|
||||
@@ -2,6 +2,7 @@ import { appendFile, mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { afterEach, beforeEach, expect, it, vi } from 'vitest'
|
||||
import SyncDatabase from '../sqlite/sync-database'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { SessionSearchService } from './session-search-service'
|
||||
import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture'
|
||||
@@ -21,10 +22,12 @@ import {
|
||||
} from './session-search-transcript-fixtures'
|
||||
let root: string
|
||||
let store: SessionSearchStore
|
||||
let databasePath: string
|
||||
beforeEach(async () => {
|
||||
resetSessionParseCacheForTests()
|
||||
root = await mkdtemp(join(tmpdir(), 'ss-capture-audit-'))
|
||||
store = new SessionSearchStore(join(root, 'index.sqlite'))
|
||||
databasePath = join(root, 'index.sqlite')
|
||||
store = new SessionSearchStore(databasePath)
|
||||
registerSessionSearchIndexSink(store)
|
||||
})
|
||||
afterEach(async () => {
|
||||
@@ -61,15 +64,36 @@ it('refreshes indexed Codex metadata when the title index changes', async () =>
|
||||
expect(listed?.title).toBe('Renamed synthetic title')
|
||||
expect(store.search({ query: 'needle' }).hits[0]?.title).toBe('Renamed synthetic title')
|
||||
})
|
||||
it('does not rewrite indexed Codex metadata when the refresh finds no change', async () => {
|
||||
const path = join(root, CODEX_ROLLOUT_FILE)
|
||||
await writeFile(
|
||||
path,
|
||||
`${codexRolloutLines(['echo'], 'synthetic output', 'synthetic needle').join('\n')}\n`
|
||||
)
|
||||
const candidate = await sessionCandidate('codex', path, root)
|
||||
await parseAgentSessionFileCached(candidate, process.platform)
|
||||
// A cache hit still re-reads the Codex title index; an unchanged read must
|
||||
// not issue an UPDATE, because list scans repeat every few seconds.
|
||||
const update = vi.spyOn(store, 'updateMetadata')
|
||||
await parseAgentSessionFileCached(candidate, process.platform)
|
||||
expect(update).not.toHaveBeenCalled()
|
||||
})
|
||||
it('redacts credential-shaped content copied into session titles', async () => {
|
||||
const fakeKey = `sk-${'x'.repeat(40)}`
|
||||
const path = join(root, `${CLAUDE_SESSION_ID}.jsonl`)
|
||||
await writeFile(path, `${userRecord(0, `synthetic needle ${fakeKey}`)}\n`)
|
||||
await parseAgentSessionFileCached(await sessionCandidate('claude', path), process.platform)
|
||||
const body = store.db.prepare('SELECT user_text FROM messages_fts').get() as { user_text: string }
|
||||
expect(body.user_text).not.toContain(fakeKey)
|
||||
const row = store.db.prepare('SELECT title FROM sessions').get() as { title: string }
|
||||
expect(row.title).not.toContain(fakeKey)
|
||||
const reader = new SyncDatabase(databasePath, { readonly: true })
|
||||
try {
|
||||
const body = reader.prepare('SELECT user_text FROM messages_fts').get() as {
|
||||
user_text: string
|
||||
}
|
||||
expect(body.user_text).not.toContain(fakeKey)
|
||||
const row = reader.prepare('SELECT title FROM sessions').get() as { title: string }
|
||||
expect(row.title).not.toContain(fakeKey)
|
||||
} finally {
|
||||
reader.close()
|
||||
}
|
||||
})
|
||||
it('does not log malformed JSON transcript excerpts during backfill', async () => {
|
||||
const roots = isolatedScanRoots(root)
|
||||
@@ -92,6 +116,6 @@ it('does not log malformed JSON transcript excerpts during backfill', async () =
|
||||
.join(' ')
|
||||
expect(logged).not.toContain('private_sy')
|
||||
} finally {
|
||||
service.dispose()
|
||||
await service.close()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -54,6 +54,7 @@ import type * as ParseCachePersistence from '../ai-vault/session-parse-cache-per
|
||||
import type * as SourceDiscovery from '../ai-vault/session-scanner-source-discovery'
|
||||
import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache'
|
||||
import { isolatedScanRoots, jsonLines } from '../ai-vault/session-scanner-test-fixtures'
|
||||
import * as sourcePresence from './session-search-source-presence'
|
||||
import { SessionSearchService, type SessionSearchScanRoots } from './session-search-service'
|
||||
|
||||
let tempRoots: string[] = []
|
||||
@@ -66,9 +67,8 @@ beforeEach(() => {
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
for (const service of services) {
|
||||
service.dispose()
|
||||
}
|
||||
vi.restoreAllMocks()
|
||||
await Promise.all(services.map((service) => service.close()))
|
||||
services = []
|
||||
await Promise.all(tempRoots.map((root) => rm(root, { recursive: true, force: true })))
|
||||
tempRoots = []
|
||||
@@ -345,6 +345,74 @@ describe('SessionSearchService consent gate', () => {
|
||||
).toEqual(['recent-session'])
|
||||
})
|
||||
|
||||
it('answers with no hits when a disable closes the index between query rounds', async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'closed-session', 'the vacuum quota never settles')
|
||||
const service = makeService(databasePath, { enabled: true, historyDays: null })
|
||||
await service.ensureBackfill(roots)
|
||||
|
||||
// The presence pass awaits a stat between rounds; a disable landing there
|
||||
// used to leave the resumed round querying a closed database handle.
|
||||
let indexedHits = 0
|
||||
vi.spyOn(sourcePresence, 'searchPresentSessionSources').mockImplementation(
|
||||
async (args, search) => {
|
||||
indexedHits = search(args).hits.length
|
||||
await service.configure({ enabled: false, historyDays: null }, roots)
|
||||
return search(args)
|
||||
}
|
||||
)
|
||||
|
||||
const result = await service.search({ query: 'vacuum', refresh: false }, roots)
|
||||
|
||||
expect(indexedHits).toBeGreaterThan(0)
|
||||
expect(result.hits).toEqual([])
|
||||
})
|
||||
|
||||
it('waits for a parked backfill before the shutdown drops the sink', async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'draining-session', 'the vacuum quota never settles')
|
||||
const service = makeService(databasePath, { enabled: true, historyDays: null })
|
||||
let releaseBackfill!: () => void
|
||||
holdNextParseCacheLoad = new Promise<void>((resolve) => {
|
||||
releaseBackfill = resolve
|
||||
})
|
||||
const backfill = service.ensureBackfill(roots)
|
||||
await vi.waitFor(() => expect(parseCacheLoads).toBe(1))
|
||||
|
||||
// A parse parked in the backfill must finish before the store closes under
|
||||
// it, so the sink survives until the drain completes.
|
||||
let closed = false
|
||||
const closing = service.close().then(() => {
|
||||
closed = true
|
||||
})
|
||||
await new Promise((resolve) => setImmediate(resolve))
|
||||
expect(closed).toBe(false)
|
||||
expect(getSessionSearchIndexSink()).not.toBeNull()
|
||||
|
||||
releaseBackfill()
|
||||
await closing
|
||||
await backfill
|
||||
expect(getSessionSearchIndexSink()).toBeNull()
|
||||
})
|
||||
|
||||
it("leaves a replacement service's sink registered when the old one closes late", async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'handover-session', 'the vacuum quota never settles')
|
||||
const retiring = makeService(databasePath, { enabled: true, historyDays: null })
|
||||
await retiring.ensureBackfill(roots)
|
||||
|
||||
// Shutdown is async, so a replacement can claim the process-global sink
|
||||
// first; the late close must not clear a sink it no longer owns.
|
||||
const replacement = makeService(databasePath, { enabled: true, historyDays: null })
|
||||
await retiring.close()
|
||||
|
||||
expect(getSessionSearchIndexSink()).not.toBeNull()
|
||||
await replacement.ensureBackfill(roots)
|
||||
expect(
|
||||
(await replacement.search({ query: 'vacuum', refresh: false }, roots)).hits.length
|
||||
).toBeGreaterThan(0)
|
||||
})
|
||||
|
||||
it('skips transcripts older than the history bound', async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'recent-session', 'the vacuum quota never settles', 1)
|
||||
|
||||
@@ -2,7 +2,9 @@ import { mkdtemp, rm, stat, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { afterEach, expect, it, vi } from 'vitest'
|
||||
import SyncDatabase from '../sqlite/sync-database'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { openSessionSearchDatabase } from './session-search-schema'
|
||||
import { parseSearchCandidates } from './session-search-parse-candidates'
|
||||
import {
|
||||
userRecord,
|
||||
@@ -23,10 +25,20 @@ import type { SessionFileCandidate } from '../ai-vault/session-scanner-types'
|
||||
|
||||
vi.mock('./session-search-backfill-pacing', () => ({ pauseBackfill: async () => {} }))
|
||||
let directory: string | undefined
|
||||
let databasePath = ''
|
||||
let store: SessionSearchStore | undefined
|
||||
let reader: SyncDatabase | undefined
|
||||
|
||||
/** The store keeps its connection private; row assertions read the same file separately. */
|
||||
function rows(sql: string, ...values: unknown[]): unknown[] {
|
||||
reader ??= new SyncDatabase(databasePath, { readonly: true })
|
||||
return reader.prepare(sql).all(...(values as never[]))
|
||||
}
|
||||
|
||||
afterEach(async () => {
|
||||
registerSessionSearchIndexSink(null)
|
||||
reader?.close()
|
||||
reader = undefined
|
||||
store?.close()
|
||||
resetSessionParseCacheForTests()
|
||||
resetCodexSessionIndexTitleCacheForTests()
|
||||
@@ -43,7 +55,7 @@ async function fixture() {
|
||||
`${userRecord(0, 'ordinary title')}\n${assistantRecord(1, 'durableneedle')}\n`
|
||||
)
|
||||
const candidate = await candidateAt(path)
|
||||
const databasePath = join(directory, 'index.sqlite')
|
||||
databasePath = join(directory, 'index.sqlite')
|
||||
store = new SessionSearchStore(databasePath)
|
||||
registerSessionSearchIndexSink(store)
|
||||
await parseSearchCandidates(store, [candidate])
|
||||
@@ -73,14 +85,11 @@ async function candidateAt(path: string): Promise<SessionFileCandidate> {
|
||||
it('reuses an unchanged reopened index without a preview cache or replacement write', async () => {
|
||||
const { candidate, store } = await fixture()
|
||||
const apply = vi.spyOn(store, 'apply')
|
||||
const before = store.db
|
||||
.prepare('SELECT session_row_id FROM files WHERE path=?')
|
||||
.get(candidate.file.path)
|
||||
const fileRow = 'SELECT session_row_id FROM files WHERE path=?'
|
||||
const before = rows(fileRow, candidate.file.path)
|
||||
await parseSearchCandidates(store, [candidate])
|
||||
expect(apply).not.toHaveBeenCalled()
|
||||
expect(
|
||||
store.db.prepare('SELECT session_row_id FROM files WHERE path=?').get(candidate.file.path)
|
||||
).toEqual(before)
|
||||
expect(rows(fileRow, candidate.file.path)).toEqual(before)
|
||||
expect(store.search({ query: 'durableneedle' }).hits).toHaveLength(1)
|
||||
// Listing still needs its preview, even when backfill can reuse the index.
|
||||
expect(
|
||||
@@ -102,7 +111,10 @@ it('does not skip a changed or atomically replaced transcript', async () => {
|
||||
})
|
||||
|
||||
it('keeps durable reuse beyond the 4096-entry preview cache', async () => {
|
||||
store = new SessionSearchStore(':memory:')
|
||||
directory = await mkdtemp(join(tmpdir(), 'search-durable-cache-'))
|
||||
databasePath = join(directory, 'index.sqlite')
|
||||
const seedConnection = openSessionSearchDatabase(databasePath)
|
||||
store = new SessionSearchStore(databasePath)
|
||||
registerSessionSearchIndexSink(store)
|
||||
const now = Date.now()
|
||||
const candidates: SessionFileCandidate[] = Array.from({ length: 4097 }, (_, i) => ({
|
||||
@@ -117,14 +129,15 @@ it('keeps durable reuse beyond the 4096-entry preview cache', async () => {
|
||||
ino: i + 1
|
||||
}
|
||||
}))
|
||||
store.db.exec('BEGIN')
|
||||
const insert = store.db.prepare(
|
||||
seedConnection.exec('BEGIN')
|
||||
const insert = seedConnection.prepare(
|
||||
'INSERT INTO files(path,dev,ino,mtime_ms,size_bytes,byte_offset) VALUES(?,?,?,?,?,?)'
|
||||
)
|
||||
for (const { file } of candidates) {
|
||||
insert.run(file.path, file.dev!, file.ino!, now, 1, 1)
|
||||
}
|
||||
store.db.exec('COMMIT')
|
||||
seedConnection.exec('COMMIT')
|
||||
seedConnection.close()
|
||||
seedSessionParseCache(
|
||||
candidates.map(({ file }) => [
|
||||
file.path,
|
||||
@@ -142,11 +155,14 @@ it('refreshes external Codex titles on cold reuse without replacing transcript r
|
||||
const path = join(directory, CODEX_ROLLOUT_FILE)
|
||||
await writeFile(path, `${codexRolloutLines(['echo'], 'output', 'titleneedle').join('\n')}\n`)
|
||||
const candidate = await sessionCandidate('codex', path, directory)
|
||||
const databasePath = join(directory, 'index.sqlite')
|
||||
databasePath = join(directory, 'index.sqlite')
|
||||
store = new SessionSearchStore(databasePath)
|
||||
registerSessionSearchIndexSink(store)
|
||||
await parseSearchCandidates(store, [candidate])
|
||||
const before = store.db.prepare('SELECT id FROM sessions WHERE index_ready = 1').all()
|
||||
const readyRows = 'SELECT id FROM sessions WHERE index_ready = 1'
|
||||
const before = rows(readyRows)
|
||||
reader?.close()
|
||||
reader = undefined
|
||||
store.close()
|
||||
resetSessionParseCacheForTests()
|
||||
resetCodexSessionIndexTitleCacheForTests()
|
||||
@@ -163,5 +179,5 @@ it('refreshes external Codex titles on cold reuse without replacing transcript r
|
||||
await parseSearchCandidates(store, [candidate])
|
||||
expect(store.search({ query: 'titleneedle' }).hits[0].title).toBe('Renamed durable title')
|
||||
expect(apply).not.toHaveBeenCalled()
|
||||
expect(store.db.prepare('SELECT id FROM sessions WHERE index_ready = 1').all()).toEqual(before)
|
||||
expect(rows(readyRows)).toEqual(before)
|
||||
})
|
||||
|
||||
@@ -4,6 +4,7 @@ import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
applyAiVaultSearchSettings,
|
||||
applyAiVaultSearchSettingsChange,
|
||||
clearAiVaultSearchIndex,
|
||||
installAiVaultSearchSettingsSource,
|
||||
readAiVaultSearchIndexStatus
|
||||
@@ -213,3 +214,68 @@ it('requires desktop clear to retry a failed policy flush before reporting appli
|
||||
await clearAiVaultSearchIndex(persist)
|
||||
expect(readAiVaultSearchIndexStatus().applied).toBe(true)
|
||||
})
|
||||
|
||||
describe('applyAiVaultSearchSettingsChange', () => {
|
||||
it('does not reconfigure the scanner when the saved policy is unchanged', async () => {
|
||||
initSessionSearchPaths(await makeUserDataDir())
|
||||
const settings = { aiVaultSearch: { enabled: true, historyDays: 90 } }
|
||||
const persist = vi.fn()
|
||||
|
||||
applyAiVaultSearchSettingsChange(
|
||||
settings,
|
||||
{ aiVaultSearch: { ...settings.aiVaultSearch } },
|
||||
persist
|
||||
)
|
||||
// The apply chain is shared and serialized, so awaiting a later apply proves
|
||||
// the unchanged write never queued one of its own.
|
||||
await applyAiVaultSearchSettings({ aiVaultSearch: { enabled: false, historyDays: 90 } })
|
||||
|
||||
expect(configureAiVaultSearch).toHaveBeenCalledTimes(1)
|
||||
expect(configureAiVaultSearch).toHaveBeenCalledWith(
|
||||
expect.objectContaining({ enabled: false }),
|
||||
expect.anything()
|
||||
)
|
||||
expect(persist).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('forwards a pause that leaves consent and retention alone', async () => {
|
||||
initSessionSearchPaths(await makeUserDataDir())
|
||||
applyAiVaultSearchSettingsChange(
|
||||
{ aiVaultSearch: { enabled: true, historyDays: 90 } },
|
||||
{ aiVaultSearch: { enabled: true, historyDays: 90, paused: true } },
|
||||
() => undefined
|
||||
)
|
||||
|
||||
await vi.waitFor(() =>
|
||||
expect(configureAiVaultSearch).toHaveBeenCalledWith(
|
||||
expect.objectContaining({ paused: true }),
|
||||
expect.anything()
|
||||
)
|
||||
)
|
||||
})
|
||||
|
||||
it('reports a failed apply through the index status instead of throwing at the caller', async () => {
|
||||
initSessionSearchPaths(await makeUserDataDir())
|
||||
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
|
||||
configureAiVaultSearch.mockRejectedValueOnce(new Error('scanner unavailable'))
|
||||
try {
|
||||
applyAiVaultSearchSettingsChange(
|
||||
{ aiVaultSearch: { enabled: false, historyDays: null } },
|
||||
{ aiVaultSearch: { enabled: true, historyDays: null } },
|
||||
() => undefined
|
||||
)
|
||||
await vi.waitFor(() => expect(readAiVaultSearchIndexStatus().applied).toBe(false))
|
||||
expect(readAiVaultSearchIndexStatus().reason).toContain('failed or is pending')
|
||||
// The scanner failure must have been absorbed, not left for the caller.
|
||||
await vi.waitFor(() =>
|
||||
expect(warn).toHaveBeenCalledWith(
|
||||
'[settings] failed to apply agent session search settings:',
|
||||
expect.any(Error)
|
||||
)
|
||||
)
|
||||
} finally {
|
||||
warn.mockRestore()
|
||||
}
|
||||
await applyAiVaultSearchSettings({ aiVaultSearch: { enabled: true, historyDays: null } })
|
||||
})
|
||||
})
|
||||
|
||||
@@ -62,6 +62,31 @@ let applyChain: Promise<unknown> = Promise.resolve()
|
||||
let policyApplied = true
|
||||
let applyGeneration = 0
|
||||
|
||||
/**
|
||||
* Reconciles a settings write. An unchanged policy is not forwarded, so re-saving
|
||||
* the same value never restarts a running backfill, and a scanner that cannot
|
||||
* apply must not fail or delay the settings save — `readAiVaultSearchIndexStatus`
|
||||
* reports that through `applied` and `reason`.
|
||||
*/
|
||||
export function applyAiVaultSearchSettingsChange(
|
||||
before: Pick<GlobalSettings, 'aiVaultSearch'>,
|
||||
after: Pick<GlobalSettings, 'aiVaultSearch'>,
|
||||
persist: () => void | Promise<void>
|
||||
): void {
|
||||
const previous = resolveAiVaultSearchSettings(before)
|
||||
const next = resolveAiVaultSearchSettings(after)
|
||||
if (
|
||||
previous.enabled === next.enabled &&
|
||||
previous.historyDays === next.historyDays &&
|
||||
(previous.paused ?? false) === (next.paused ?? false)
|
||||
) {
|
||||
return
|
||||
}
|
||||
void applyAiVaultSearchSettings(after, { persist }).catch((error: unknown) => {
|
||||
console.warn('[settings] failed to apply agent session search settings:', error)
|
||||
})
|
||||
}
|
||||
|
||||
export function readAiVaultSearchIndexStatus(): AiVaultSearchIndexStatus {
|
||||
const capability = sessionSearchCapability()
|
||||
const available =
|
||||
|
||||
@@ -6,6 +6,9 @@ import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache'
|
||||
import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture'
|
||||
import { indexTokens } from './session-search-query-planner'
|
||||
import { sessionRowFilter } from './session-search-row-filter'
|
||||
import { splitAiVaultSearchQuery } from '../../shared/ai-vault-search-query-operators'
|
||||
import { openSessionSearchDatabase } from './session-search-schema'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { parseTranscript as parse, userRecord } from './session-search-transcript-fixtures'
|
||||
@@ -233,3 +236,43 @@ describe('SessionSearchStore.search snippets', () => {
|
||||
store.close()
|
||||
})
|
||||
})
|
||||
|
||||
describe('the planner tokenizer draws the same boundaries as unicode61', () => {
|
||||
// unicode61 folds case and strips Latin diacritics on both index and query side.
|
||||
function asIndexed(token: string): string {
|
||||
return token.toLowerCase().normalize('NFD').replaceAll(/\p{M}/gu, '')
|
||||
}
|
||||
|
||||
it('produces exactly the terms fts5vocab reports for the same text', async () => {
|
||||
const db = await openDatabase()
|
||||
const corpus =
|
||||
'resolveTerminalPath src/main/foo-bar.ts a.b C++ #123 修复 café naïve MAX_TOKEN x'
|
||||
insertMessageRow(db, FIRST_ROWID, corpus)
|
||||
const indexed = (
|
||||
db.prepare('SELECT term FROM messages_vocab ORDER BY term').all() as { term: string }[]
|
||||
).map((row) => row.term)
|
||||
|
||||
expect([...new Set(indexTokens(corpus).map(asIndexed))].sort()).toEqual(indexed)
|
||||
})
|
||||
})
|
||||
|
||||
describe('a cwd scope seeks the cwd_key index instead of scanning it', () => {
|
||||
it('plans the scope condition as a SEARCH on sessions_cwd_key', async () => {
|
||||
const db = await openDatabase()
|
||||
const filter = sessionRowFilter(
|
||||
{ query: 'needle', scopePaths: ['/work/app'] },
|
||||
splitAiVaultSearchQuery('needle')
|
||||
)
|
||||
const plan = (
|
||||
db
|
||||
.prepare(
|
||||
`EXPLAIN QUERY PLAN SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')}`
|
||||
)
|
||||
.all(...filter.values) as { detail: string }[]
|
||||
).map((row) => row.detail)
|
||||
|
||||
expect(plan.join(' | ')).toContain('sessions_cwd_key')
|
||||
expect(plan.some((detail) => detail.startsWith('SEARCH'))).toBe(true)
|
||||
expect(plan.some((detail) => detail.startsWith('SCAN sessions'))).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
import type { AiVaultAgent } from '../../shared/ai-vault-types'
|
||||
import type { AiVaultSearchArgs, AiVaultSearchHit } from '../../shared/ai-vault-search-types'
|
||||
import {
|
||||
AI_VAULT_SEARCH_LIMIT_DEFAULT,
|
||||
AI_VAULT_SEARCH_LIMIT_MAX
|
||||
} from '../../shared/ai-vault-search-types'
|
||||
import { isCollapsibleContentHash } from './session-search-content-hash'
|
||||
|
||||
// Subtracted per session: `0.02 · ln(1 + messages)`; slightly positive on both eval sets.
|
||||
const LENGTH_PRIOR = 0.02
|
||||
|
||||
export type SessionRow = {
|
||||
id: number
|
||||
agent: AiVaultAgent
|
||||
session_id: string
|
||||
file_path: string
|
||||
codex_home: string | null
|
||||
title: string
|
||||
cwd: string | null
|
||||
branch: string | null
|
||||
updated_at: string | null
|
||||
message_count: number
|
||||
resume_command: string
|
||||
content_hash: string | null
|
||||
content_hash_count: number
|
||||
}
|
||||
|
||||
/** The one message that stands for a session: its best-scoring match. */
|
||||
export type MessageRow = {
|
||||
rowid: number
|
||||
score: number
|
||||
session_row_id: number
|
||||
role: string
|
||||
ts: string | null
|
||||
}
|
||||
|
||||
type ScoredSession = {
|
||||
session: SessionRow
|
||||
message: MessageRow
|
||||
score: number
|
||||
duplicateCount: number
|
||||
}
|
||||
|
||||
export function sessionFields(session: SessionRow): Omit<AiVaultSearchHit, 'score' | 'evidence'> {
|
||||
return {
|
||||
agent: session.agent,
|
||||
sessionId: session.session_id,
|
||||
filePath: session.file_path,
|
||||
codexHome: session.codex_home,
|
||||
title: session.title,
|
||||
cwd: session.cwd,
|
||||
branch: session.branch,
|
||||
updatedAt: session.updated_at,
|
||||
messageCount: session.message_count,
|
||||
resumeCommand: session.resume_command
|
||||
}
|
||||
}
|
||||
|
||||
// Why: the desktop IPC forwards its payload unvalidated, so a non-positive
|
||||
// limit must be clamped here or `LIMIT -1` / `slice(0, -1)` leak through.
|
||||
export function resolveLimit(args: AiVaultSearchArgs): number {
|
||||
const requested = Number.isInteger(args.limit)
|
||||
? (args.limit as number)
|
||||
: AI_VAULT_SEARCH_LIMIT_DEFAULT
|
||||
return Math.min(Math.max(1, requested), AI_VAULT_SEARCH_LIMIT_MAX)
|
||||
}
|
||||
|
||||
/**
|
||||
* Everything between "these sessions matched" and "this is the page": the length
|
||||
* prior, fork folding, the caller's order, and the cut. Retrieval stays in SQL;
|
||||
* nothing here touches the database, and `snippet` runs only for the page.
|
||||
*/
|
||||
export function rankSessionHits(
|
||||
sessions: readonly SessionRow[],
|
||||
matches: ReadonlyMap<number, MessageRow>,
|
||||
args: AiVaultSearchArgs,
|
||||
snippet: (message: MessageRow) => string
|
||||
): AiVaultSearchHit[] {
|
||||
const scored = collapseForks(
|
||||
sessions.map((session) => {
|
||||
const message = matches.get(session.id)!
|
||||
return {
|
||||
session,
|
||||
message,
|
||||
score: message.score - LENGTH_PRIOR * Math.log(1 + session.message_count),
|
||||
duplicateCount: 1
|
||||
}
|
||||
})
|
||||
)
|
||||
scored.sort((left, right) =>
|
||||
args.sort === 'newest'
|
||||
? (right.session.updated_at ?? '').localeCompare(left.session.updated_at ?? '')
|
||||
: right.score - left.score
|
||||
)
|
||||
return scored.slice(0, resolveLimit(args)).map(({ session, message, score, duplicateCount }) => ({
|
||||
...sessionFields(session),
|
||||
score,
|
||||
...(duplicateCount > 1 ? { duplicateCount } : {}),
|
||||
evidence: {
|
||||
role: message.role as AiVaultSearchHit['evidence']['role'],
|
||||
timestamp: message.ts,
|
||||
snippet: snippet(message)
|
||||
}
|
||||
}))
|
||||
}
|
||||
|
||||
/**
|
||||
* Folds forked copies of one conversation into a single hit: same opening
|
||||
* prefix, newest `updated_at` wins, the rest become `duplicateCount`. Done here
|
||||
* and not at write time so index rows stay per file (cursors and deletes).
|
||||
*/
|
||||
function collapseForks(scored: ScoredSession[]): ScoredSession[] {
|
||||
const groups = new Map<string, ScoredSession[]>()
|
||||
for (const entry of scored) {
|
||||
const { content_hash: hash, content_hash_count: count, id } = entry.session
|
||||
const key = isCollapsibleContentHash(hash, count) ? `hash:${hash}` : `session:${id}`
|
||||
const group = groups.get(key)
|
||||
if (group) {
|
||||
group.push(entry)
|
||||
} else {
|
||||
groups.set(key, [entry])
|
||||
}
|
||||
}
|
||||
const collapsed: ScoredSession[] = []
|
||||
for (const group of groups.values()) {
|
||||
if (group.length === 1) {
|
||||
collapsed.push(group[0]!)
|
||||
continue
|
||||
}
|
||||
const winner = group.reduce((best, entry) => (isNewer(entry, best) ? entry : best))
|
||||
collapsed.push({ ...winner, duplicateCount: group.length })
|
||||
}
|
||||
return collapsed
|
||||
}
|
||||
|
||||
function isNewer(entry: ScoredSession, best: ScoredSession): boolean {
|
||||
const order = (entry.session.updated_at ?? '').localeCompare(best.session.updated_at ?? '')
|
||||
return order === 0 ? entry.score > best.score : order > 0
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
|
||||
const COMPACT_PAGES_PER_STEP = 2000
|
||||
|
||||
/** Hands freed pages back to the filesystem in bounded steps, never one long stall. */
|
||||
export async function compactSessionSearchIndex(
|
||||
db: SyncDatabase,
|
||||
stopped: () => boolean
|
||||
): Promise<void> {
|
||||
let freed = Number(db.pragma('freelist_count', { simple: true }))
|
||||
while (!stopped() && freed > 0) {
|
||||
db.pragma(`incremental_vacuum(${COMPACT_PAGES_PER_STEP})`)
|
||||
const remaining = Number(db.pragma('freelist_count', { simple: true }))
|
||||
// Why: without auto_vacuum the step is a no-op; never spin on it.
|
||||
if (remaining >= freed) {
|
||||
return
|
||||
}
|
||||
freed = remaining
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
import { assertSearchWalBudget } from './session-search-wal-budget'
|
||||
import { assertSearchWalBudget, SEARCH_WAL_PENDING_BYTES } from './session-search-wal-budget'
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import type { AiVaultSession } from '../../shared/ai-vault-types'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
@@ -11,13 +11,26 @@ import type {
|
||||
import { EMPTY_CONTENT_HASH, foldContentHash } from './session-search-content-hash'
|
||||
import { SessionSearchFileRecords } from './session-search-file-records'
|
||||
import { insertSearchMessage, searchMessageRows } from './session-search-message-rows'
|
||||
import { discardSearchBatch, retireSearchSession } from './session-search-write-recovery'
|
||||
import { discardSearchBatch, retireSearchSession } from './session-search-pending-deletes'
|
||||
import { redactSessionSearchText } from './session-search-redaction'
|
||||
import { sessionSearchPathKey } from './session-search-path-key'
|
||||
export { chunkMessageText } from './session-search-message-rows'
|
||||
|
||||
export const SEARCH_WRITE_ROWS_PER_STEP = 128
|
||||
export const SEARCH_WRITE_CHARS_PER_STEP = 256 * 1024
|
||||
// Why sampled: the checkpoint costs more than the step it guards, and the backlog it
|
||||
// watches only grows while a second connection pins a snapshot, which takes seconds.
|
||||
const WAL_BUDGET_EVERY_STEPS = 16
|
||||
|
||||
export type SessionSearchApplyOptions = {
|
||||
/** False once this write is superseded; staging stops without publishing. */
|
||||
active?: () => boolean
|
||||
yieldStep?: () => Promise<void>
|
||||
/** False once the database is gone; gates the discard tombstone. Defaults to `active`. */
|
||||
available?: () => boolean
|
||||
}
|
||||
|
||||
type ResolvedApplyOptions = Required<SessionSearchApplyOptions>
|
||||
|
||||
export type SessionSearchMetadata = Pick<
|
||||
AiVaultSession,
|
||||
@@ -38,7 +51,10 @@ export class SessionSearchIndexWriter {
|
||||
private activePath: string | null = null
|
||||
private invalidated = false
|
||||
private pending: Promise<unknown> = Promise.resolve()
|
||||
constructor(private readonly db: SyncDatabase) {
|
||||
constructor(
|
||||
private readonly db: SyncDatabase,
|
||||
private readonly walBudgetBytes: number = SEARCH_WAL_PENDING_BYTES
|
||||
) {
|
||||
this.records = new SessionSearchFileRecords(db)
|
||||
}
|
||||
indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null {
|
||||
@@ -84,17 +100,21 @@ export class SessionSearchIndexWriter {
|
||||
|
||||
apply(
|
||||
update: SessionSearchIndexWrite,
|
||||
active: () => boolean = () => true,
|
||||
yieldStep: () => Promise<void> = yieldToEventLoop,
|
||||
available: () => boolean = active
|
||||
options: SessionSearchApplyOptions = {}
|
||||
): Promise<boolean> {
|
||||
const active = options.active ?? (() => true)
|
||||
const resolved: ResolvedApplyOptions = {
|
||||
active,
|
||||
yieldStep: options.yieldStep ?? yieldToEventLoop,
|
||||
available: options.available ?? active
|
||||
}
|
||||
const run = this.pending
|
||||
.catch(() => undefined)
|
||||
.then(async () => {
|
||||
this.activePath = update.candidate.file.path
|
||||
this.invalidated = false
|
||||
try {
|
||||
return await this.stage(update, active, yieldStep, available)
|
||||
return await this.stage(update, resolved)
|
||||
} finally {
|
||||
this.activePath = null
|
||||
}
|
||||
@@ -130,14 +150,11 @@ export class SessionSearchIndexWriter {
|
||||
|
||||
private async stage(
|
||||
update: SessionSearchIndexWrite,
|
||||
active: () => boolean,
|
||||
yieldStep: () => Promise<void>,
|
||||
available: () => boolean
|
||||
{ active, yieldStep, available }: ResolvedApplyOptions
|
||||
): Promise<boolean> {
|
||||
if (!active()) {
|
||||
return false
|
||||
}
|
||||
assertSearchWalBudget(this.db)
|
||||
const path = update.candidate.file.path
|
||||
const existing = this.file(path)
|
||||
const append =
|
||||
@@ -183,6 +200,7 @@ export class SessionSearchIndexWriter {
|
||||
}
|
||||
const rows = capturedRows()
|
||||
let next = await rows.next()
|
||||
let step = 0
|
||||
while (!next.done) {
|
||||
const batch: SessionSearchCapturedMessage[] = []
|
||||
let chars = 0
|
||||
@@ -199,7 +217,9 @@ export class SessionSearchIndexWriter {
|
||||
await rows.return(undefined)
|
||||
return false
|
||||
}
|
||||
assertSearchWalBudget(this.db)
|
||||
if (step++ % WAL_BUDGET_EVERY_STEPS === 0) {
|
||||
assertSearchWalBudget(this.db, this.walBudgetBytes)
|
||||
}
|
||||
this.db.exec('BEGIN IMMEDIATE')
|
||||
try {
|
||||
for (const message of batch) {
|
||||
@@ -212,7 +232,7 @@ export class SessionSearchIndexWriter {
|
||||
}
|
||||
await yieldStep()
|
||||
}
|
||||
const result = 'result' in update ? await update.result : update
|
||||
const result = await update.result
|
||||
if (!active() || !unchanged()) {
|
||||
return false
|
||||
}
|
||||
@@ -231,7 +251,10 @@ export class SessionSearchIndexWriter {
|
||||
retireSearchSession(this.db, existing.session_row_id)
|
||||
}
|
||||
this.db.prepare('UPDATE sessions SET index_ready=1 WHERE id=?').run(sessionId)
|
||||
this.db.prepare('UPDATE search_write_batches SET published=1 WHERE id=?').run(batchId)
|
||||
// Clearing the pointer before dropping the batch is what makes a recycled
|
||||
// rowid harmless: no published row can name a later in-flight batch.
|
||||
this.db.prepare('UPDATE messages SET batch_id=NULL WHERE batch_id=?').run(batchId)
|
||||
this.db.prepare('DELETE FROM search_write_batches WHERE id=?').run(batchId)
|
||||
this.records.upsertFile(update.candidate, result.byteOffset, sessionId)
|
||||
this.db.exec('COMMIT')
|
||||
} catch (error) {
|
||||
@@ -240,13 +263,12 @@ export class SessionSearchIndexWriter {
|
||||
}
|
||||
return true
|
||||
} finally {
|
||||
if (available()) {
|
||||
const batch = this.db
|
||||
.prepare('SELECT published FROM search_write_batches WHERE id=?')
|
||||
.get(batchId) as { published: number } | undefined
|
||||
if (batch?.published === 0) {
|
||||
discardSearchBatch(this.db, sessionId, batchId, !append)
|
||||
}
|
||||
// A surviving batch row means publish never ran, whatever ended the stage.
|
||||
if (
|
||||
available() &&
|
||||
this.db.prepare('SELECT 1 FROM search_write_batches WHERE id=?').get(batchId)
|
||||
) {
|
||||
discardSearchBatch(this.db, sessionId, batchId, !append)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -41,6 +41,24 @@ it('groups adjacent live writes into one burst and keeps failures visible', () =
|
||||
expect(progress.snapshot()).toMatchObject({ phase: 'updating', filesTotal: 2, startedAt })
|
||||
progress.writeFailed()
|
||||
second()
|
||||
expect(progress.snapshot()).toMatchObject({ phase: 'error', failures: 1 })
|
||||
})
|
||||
|
||||
it('lets a later successful write supersede a write error, but not a failed backfill', () => {
|
||||
const progress = new SessionSearchIndexingProgress()
|
||||
const failing = progress.beginWrite()
|
||||
progress.writeFailed()
|
||||
failing()
|
||||
expect(progress.snapshot().phase).toBe('error')
|
||||
progress.beginWrite()()
|
||||
expect(progress.snapshot().phase).toBe('complete')
|
||||
|
||||
// A failed backfill stays red: configure() reads this phase to decide whether
|
||||
// to drop the memoized pass and enumerate again.
|
||||
progress.discover()
|
||||
progress.discovered(1, 0)
|
||||
progress.processed(true)
|
||||
progress.finish()
|
||||
expect(progress.snapshot().phase).toBe('error')
|
||||
progress.beginWrite()()
|
||||
expect(progress.snapshot().phase).toBe('error')
|
||||
|
||||
@@ -13,6 +13,9 @@ export class SessionSearchIndexingProgress {
|
||||
private backfilling = false
|
||||
private activeWrites = 0
|
||||
private lastWriteAt = 0
|
||||
// Why: a failed backfill keeps the badge red until it is retried, but a
|
||||
// transient write failure must not outlive the next successful write.
|
||||
private failedWrite = false
|
||||
|
||||
snapshot(): AiVaultSearchIndexingProgress {
|
||||
return { ...this.value, ...(this.paused ? { phase: 'paused' as const } : {}) }
|
||||
@@ -24,6 +27,7 @@ export class SessionSearchIndexingProgress {
|
||||
|
||||
discover(): void {
|
||||
this.backfilling = true
|
||||
this.failedWrite = false
|
||||
this.value = {
|
||||
phase: 'discovering',
|
||||
filesProcessed: 0,
|
||||
@@ -51,7 +55,7 @@ export class SessionSearchIndexingProgress {
|
||||
|
||||
beginWrite(): () => void {
|
||||
this.activeWrites++
|
||||
if (!this.backfilling && !this.paused && this.value.phase !== 'error') {
|
||||
if (!this.backfilling && !this.paused && (this.value.phase !== 'error' || this.failedWrite)) {
|
||||
if (Date.now() - this.lastWriteAt > 1000 && this.activeWrites === 1) {
|
||||
this.value = {
|
||||
phase: 'updating',
|
||||
@@ -72,6 +76,7 @@ export class SessionSearchIndexingProgress {
|
||||
this.value.filesProcessed++
|
||||
if (this.activeWrites === 0) {
|
||||
this.value.phase = 'complete'
|
||||
this.failedWrite = false
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -79,6 +84,7 @@ export class SessionSearchIndexingProgress {
|
||||
|
||||
writeFailed(): void {
|
||||
if (!this.backfilling) {
|
||||
this.failedWrite = true
|
||||
this.value.failures++
|
||||
this.value.phase = 'error'
|
||||
}
|
||||
|
||||
@@ -1,81 +0,0 @@
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { assertSearchWalBudget } from './session-search-wal-budget'
|
||||
import { redactSessionSearchText } from './session-search-redaction'
|
||||
|
||||
const COMPACT_PAGES_PER_STEP = 2000
|
||||
const WARM_ROWS_PER_STEP = 50_000
|
||||
const SEARCH_LOG_LIMIT = 5000
|
||||
|
||||
export class SessionSearchMaintenance {
|
||||
private warmed: Promise<void> | null = null
|
||||
constructor(
|
||||
private readonly db: SyncDatabase,
|
||||
private readonly closed: () => boolean,
|
||||
private readonly onError: (error: unknown) => void
|
||||
) {}
|
||||
async compact(signal?: AbortSignal): Promise<void> {
|
||||
try {
|
||||
let freed = Number(this.db.pragma('freelist_count', { simple: true }))
|
||||
while (!this.closed() && !signal?.aborted && freed > 0) {
|
||||
assertSearchWalBudget(this.db)
|
||||
this.db.pragma(`incremental_vacuum(${COMPACT_PAGES_PER_STEP})`)
|
||||
const remaining = Number(this.db.pragma('freelist_count', { simple: true }))
|
||||
// Why: without auto_vacuum the step is a no-op; never spin on it.
|
||||
if (remaining >= freed) {
|
||||
return
|
||||
}
|
||||
freed = remaining
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
} catch (error) {
|
||||
this.onError(error)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads the messages table through in slices so its pages sit in the OS
|
||||
* cache before the first query joins against it. Measured on a 4 GB index:
|
||||
* the first query after a cold start drops from ~1.3 s to ~0.45 s, and each
|
||||
* slice holds the connection for under 50 ms.
|
||||
*/
|
||||
warm(): Promise<void> {
|
||||
this.warmed ??= this.readMessagesThrough().catch((error) => this.onError(error))
|
||||
return this.warmed
|
||||
}
|
||||
|
||||
private async readMessagesThrough(): Promise<void> {
|
||||
const max = (
|
||||
this.db.prepare('SELECT max(id) AS id FROM messages').get() as { id: number | null }
|
||||
).id
|
||||
const touch = this.db.prepare(
|
||||
'SELECT count(*) FROM messages WHERE id BETWEEN ? AND ? AND role IS NOT NULL'
|
||||
)
|
||||
for (let low = 1; max !== null && low <= max; low += WARM_ROWS_PER_STEP) {
|
||||
if (this.closed()) {
|
||||
return
|
||||
}
|
||||
touch.get(low, low + WARM_ROWS_PER_STEP - 1)
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
}
|
||||
|
||||
logQuery(query: string, route: string, hits: number, durationMs: number): void {
|
||||
try {
|
||||
assertSearchWalBudget(this.db)
|
||||
this.db
|
||||
.prepare(
|
||||
'INSERT INTO search_log(ts, query, route, hits, duration_ms) VALUES (?, ?, ?, ?, ?)'
|
||||
)
|
||||
.run(new Date().toISOString(), redactSessionSearchText(query), route, hits, durationMs)
|
||||
this.db
|
||||
.prepare(
|
||||
`DELETE FROM search_log WHERE id <= (
|
||||
SELECT id FROM search_log ORDER BY id DESC LIMIT 1 OFFSET ?)`
|
||||
)
|
||||
.run(SEARCH_LOG_LIMIT)
|
||||
} catch (error) {
|
||||
this.onError(error)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -82,7 +82,7 @@ class LoopbackWorker extends EventEmitter {
|
||||
)
|
||||
.then(async (session) => {
|
||||
await capture.flush()
|
||||
this.emit('message', { id: request.id, ok: true, value: { session } })
|
||||
this.emit('message', { id: request.id, kind: 'result', value: { session } })
|
||||
})
|
||||
.catch((error) => {
|
||||
if (!this.terminated) {
|
||||
|
||||
@@ -185,6 +185,26 @@ describe('OpenCode SQLite session freshness', () => {
|
||||
expect(store.coverage().messagesIndexed).toBe(2)
|
||||
})
|
||||
|
||||
it('keeps listing a session whose part blob the capture read cannot decode', async () => {
|
||||
const dbPath = await createOpenCodeDb()
|
||||
const db = new Database(dbPath)
|
||||
// Valid JSON that is not an object, so SQLite's json_extract tolerates it in
|
||||
// the preview query and only the search capture read has to survive it.
|
||||
db.prepare(
|
||||
`INSERT INTO part (id, message_id, session_id, time_created, time_updated, data)
|
||||
VALUES ('prt_bad', 'msg_1', ?, ?, ?, 'null')`
|
||||
).run(SESSION_ID, CREATED_MS + 600, CREATED_MS + 600)
|
||||
db.close()
|
||||
|
||||
// The parse must not reject: an index failure may never cost the session its
|
||||
// place in the list, and the readable turns must still reach the index.
|
||||
await expect(refreshRecent(dbPath)).resolves.toMatchObject({ fullParses: 1 })
|
||||
expect(store.search({ query: 'vacuum quota' }).hits).toMatchObject([
|
||||
{ agent: 'opencode', sessionId: SESSION_ID }
|
||||
])
|
||||
expect(store.coverage().messagesIndexed).toBe(1)
|
||||
})
|
||||
|
||||
it('reuses the cached parse and leaves the index alone when nothing changed', async () => {
|
||||
const dbPath = await createOpenCodeDb()
|
||||
await refreshRecent(dbPath)
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
|
||||
const WARM_ROWS_PER_STEP = 50_000
|
||||
|
||||
/**
|
||||
* Reads the messages table through in slices so its pages sit in the OS cache
|
||||
* before the first query joins against it. Measured on a 4 GB index: the first
|
||||
* query after a cold start drops from ~1.3 s to ~0.45 s, and each slice holds
|
||||
* the connection for under 50 ms.
|
||||
*/
|
||||
export async function warmSessionSearchPages(
|
||||
db: SyncDatabase,
|
||||
stopped: () => boolean
|
||||
): Promise<void> {
|
||||
const max = (db.prepare('SELECT max(id) AS id FROM messages').get() as { id: number | null }).id
|
||||
const touch = db.prepare(
|
||||
'SELECT count(*) FROM messages WHERE id BETWEEN ? AND ? AND role IS NOT NULL'
|
||||
)
|
||||
for (let low = 1; max !== null && low <= max; low += WARM_ROWS_PER_STEP) {
|
||||
if (stopped()) {
|
||||
return
|
||||
}
|
||||
touch.get(low, low + WARM_ROWS_PER_STEP - 1)
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
}
|
||||
@@ -16,11 +16,16 @@ import type { SessionFileCandidate } from '../ai-vault/session-scanner-types'
|
||||
import type { SessionSearchStore } from './session-search-store'
|
||||
import { pauseBackfill } from './session-search-backfill-pacing'
|
||||
|
||||
export type ParseSearchCandidatesOptions = {
|
||||
signal?: AbortSignal
|
||||
/** Backfill only: report each file to the progress bar and yield to waiting searches. */
|
||||
onFileProcessed?: (failed: boolean) => Promise<void>
|
||||
}
|
||||
|
||||
export async function parseSearchCandidates(
|
||||
store: SessionSearchStore,
|
||||
candidates: SessionFileCandidate[],
|
||||
signal?: AbortSignal,
|
||||
waitForSearches?: () => Promise<void>
|
||||
{ signal, onFileProcessed }: ParseSearchCandidatesOptions = {}
|
||||
): Promise<void> {
|
||||
const stats = createSessionParseStats()
|
||||
let sinceYield = 0
|
||||
@@ -75,9 +80,8 @@ export async function parseSearchCandidates(
|
||||
error instanceof Error ? error.name : 'ParseError'
|
||||
)
|
||||
}
|
||||
if (waitForSearches && !signal?.aborted) {
|
||||
store.indexing.processed(failed || store.failures > failures)
|
||||
await waitForSearches()
|
||||
if (onFileProcessed && !signal?.aborted) {
|
||||
await onFileProcessed(failed || store.failures > failures)
|
||||
}
|
||||
sinceYield++
|
||||
if (sinceYield >= 8) {
|
||||
|
||||
@@ -36,7 +36,7 @@ beforeEach(async () => {
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
service.dispose()
|
||||
await service.close()
|
||||
await rm(root, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
@@ -65,7 +65,7 @@ it('pauses ordinary scanner writes and query refreshes, then catches up without
|
||||
it('keeps pause across a scanner restart and preserves saved results', async () => {
|
||||
await service.ensureBackfill(roots)
|
||||
await service.configure({ ...enabled, paused: true }, roots)
|
||||
service.dispose()
|
||||
await service.close()
|
||||
service = new SessionSearchService({ databasePath, ...enabled, paused: true })
|
||||
await appendFile(file, `${userRecord(1, 'restartneedle')}\n`)
|
||||
await service.ensureBackfill(roots)
|
||||
@@ -94,7 +94,7 @@ it('clear while paused stays paused; disabling closes the sink and retains the p
|
||||
|
||||
it('an interrupted write cannot be revived by a quick resume', async () => {
|
||||
const { SessionSearchStore } = await import('./session-search-store')
|
||||
const { stagedWriteUpdate } = await import('./session-search-staged-write-fixtures')
|
||||
const { stagedWriteUpdate } = await import('./session-search-staged-write-test-fixture')
|
||||
const store = getSessionSearchIndexSink() as InstanceType<typeof SessionSearchStore>
|
||||
await store.apply(stagedWriteUpdate('savedneedle', 1))
|
||||
let entered!: () => void
|
||||
|
||||
+7
-8
@@ -1,12 +1,10 @@
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
|
||||
export function retireSearchSession(
|
||||
db: SyncDatabase,
|
||||
sessionId: number,
|
||||
path = `\0session:${sessionId}`
|
||||
): void {
|
||||
// Why the NUL prefix: tombstones share the file-path key space with real files, and
|
||||
// `\0` cannot occur in one, so a synthetic key still gets the per-path cleanup mutex.
|
||||
export function retireSearchSession(db: SyncDatabase, sessionId: number): void {
|
||||
db.prepare('INSERT OR IGNORE INTO search_pending_deletes(path,session_row_id) VALUES (?,?)').run(
|
||||
path,
|
||||
`\0session:${sessionId}`,
|
||||
sessionId
|
||||
)
|
||||
}
|
||||
@@ -28,9 +26,10 @@ export function discardSearchBatch(
|
||||
|
||||
/** Only called on open, before this store can have active writers. */
|
||||
export function recoverSearchWrites(db: SyncDatabase): void {
|
||||
// A staging session or a batch row that outlived its writer is by definition unfinished:
|
||||
// publish clears both in the same transaction that makes the rows visible.
|
||||
db.exec(`INSERT OR IGNORE INTO search_pending_deletes(path,session_row_id)
|
||||
SELECT char(0)||'session:'||id,id FROM sessions WHERE index_ready=0;
|
||||
INSERT OR IGNORE INTO search_pending_deletes(path,session_row_id,batch_id)
|
||||
SELECT char(0)||'batch:'||b.id,b.session_row_id,b.id FROM search_write_batches b
|
||||
JOIN sessions s ON s.id=b.session_row_id WHERE b.published=0 AND s.index_ready=1`)
|
||||
SELECT char(0)||'batch:'||id,session_row_id,id FROM search_write_batches`)
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { redactSessionSearchText } from './session-search-redaction'
|
||||
|
||||
const SEARCH_LOG_LIMIT = 5000
|
||||
|
||||
/** Telemetry the eval set is rebuilt from; the query text is redacted before it lands. */
|
||||
export function logSessionSearchQuery(
|
||||
db: SyncDatabase,
|
||||
entry: { query: string; route: string; hits: number; durationMs: number }
|
||||
): void {
|
||||
db.prepare(
|
||||
'INSERT INTO search_log(ts, query, route, hits, duration_ms) VALUES (?, ?, ?, ?, ?)'
|
||||
).run(
|
||||
new Date().toISOString(),
|
||||
redactSessionSearchText(entry.query),
|
||||
entry.route,
|
||||
entry.hits,
|
||||
entry.durationMs
|
||||
)
|
||||
db.prepare(
|
||||
`DELETE FROM search_log WHERE id <= (
|
||||
SELECT id FROM search_log ORDER BY id DESC LIMIT 1 OFFSET ?)`
|
||||
).run(SEARCH_LOG_LIMIT)
|
||||
}
|
||||
@@ -10,6 +10,7 @@ import { parseTranscript, userRecord } from './session-search-transcript-fixture
|
||||
const APP_ID = 'aaaaaaaa-0000-4000-8000-00000000000a'
|
||||
const SERVICE_ID = 'aaaaaaaa-0000-4000-8000-00000000000b'
|
||||
const NEWER_APP_ID = 'aaaaaaaa-0000-4000-8000-00000000000c'
|
||||
const ACCENTED_ID = 'aaaaaaaa-0000-4000-8000-00000000000d'
|
||||
|
||||
let tempRoots: string[] = []
|
||||
let root: string
|
||||
@@ -93,6 +94,30 @@ describe('repo: and path: operators at the query level', () => {
|
||||
expect(sessionIds('repo:service', { scopePaths: ['/repo'] })).toEqual([])
|
||||
})
|
||||
|
||||
it('ORs terms within one operator key and ANDs across keys', () => {
|
||||
expect(sessionIds('harbor repo:app repo:service').sort()).toEqual(
|
||||
[APP_ID, NEWER_APP_ID, SERVICE_ID].sort()
|
||||
)
|
||||
expect(sessionIds('harbor path:/repo path:/other').sort()).toEqual(
|
||||
[APP_ID, NEWER_APP_ID, SERVICE_ID].sort()
|
||||
)
|
||||
// Two absolute paths would be unsatisfiable if same-key terms ANDed.
|
||||
expect(sessionIds('harbor path:/repo/app path:/other').sort()).toEqual(
|
||||
[APP_ID, NEWER_APP_ID, SERVICE_ID].sort()
|
||||
)
|
||||
expect(sessionIds('harbor repo:app path:/other')).toEqual([])
|
||||
})
|
||||
|
||||
it('folds a substring path term as far as SQLite can, and an absolute one not at all', async () => {
|
||||
// LIKE folds ASCII on both sides; `lower()` would fold ASCII and still miss `É`.
|
||||
await indexSession(ACCENTED_ID, '/repo/CAFÉ', 'harbor lantern')
|
||||
expect(sessionIds('harbor path:café')).toEqual([])
|
||||
expect(sessionIds('harbor path:CAFÉ')).toEqual([ACCENTED_ID])
|
||||
expect(sessionIds('harbor path:APP').sort()).toEqual([APP_ID, NEWER_APP_ID].sort())
|
||||
// An absolute term is an identity claim, so POSIX case is not folded.
|
||||
expect(sessionIds('harbor path:/REPO/app')).toEqual([])
|
||||
})
|
||||
|
||||
it('keeps operator text out of the FTS expression', () => {
|
||||
// `harbo` is one edit from the indexed `harbor`: if the operator value
|
||||
// reached the planner it would come back on a typo+ route.
|
||||
|
||||
@@ -32,13 +32,20 @@ export function isLiteralQuery(query: string): boolean {
|
||||
return QUOTED.test(query) || LITERAL_SHAPE.test(query)
|
||||
}
|
||||
|
||||
function indexTokens(query: string): string[] {
|
||||
/**
|
||||
* The tokenizer contract, unfolded: the same boundaries FTS5 draws for
|
||||
* `unicode61 tokenchars '_.-/+'`. Pinned against real `fts5vocab` output in
|
||||
* session-search-fts5-contract.test.ts, which is what makes it safe to plan a
|
||||
* query without asking SQLite.
|
||||
*/
|
||||
export function indexTokens(query: string, limit = Number.POSITIVE_INFINITY): string[] {
|
||||
const out: string[] = []
|
||||
for (const match of query.matchAll(INDEX_TOKEN)) {
|
||||
const token = match[0]
|
||||
// Separators alone (`--`, `...`) are a token to FTS5 but never a search term.
|
||||
if (/[\p{L}\p{N}\p{Co}]/u.test(token)) {
|
||||
out.push(token)
|
||||
if (out.length >= MAX_BODY_TERMS) {
|
||||
if (out.length >= limit) {
|
||||
break
|
||||
}
|
||||
}
|
||||
@@ -47,7 +54,7 @@ function indexTokens(query: string): string[] {
|
||||
}
|
||||
|
||||
export function planSessionSearchQuery(query: string): SessionSearchQueryPlan {
|
||||
const raw = indexTokens(query)
|
||||
const raw = indexTokens(query, MAX_BODY_TERMS)
|
||||
let body = isLiteralQuery(query)
|
||||
? raw
|
||||
: raw.filter((token) => !STOP_WORDS.has(token.toLowerCase()))
|
||||
|
||||
@@ -1,78 +1,93 @@
|
||||
import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import { sessionSearchPathKey } from './session-search-path-key'
|
||||
import { describe, it, expect } from 'vitest'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { SessionSearchService } from './session-search-service'
|
||||
import { openSessionSearchIndexFile } from './session-search-staged-write-test-fixture'
|
||||
import { isolatedScanRoots } from '../ai-vault/session-scanner-test-fixtures'
|
||||
import { mkdir, mkdtemp, writeFile, utimes } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { userRecord, parseTranscript } from './session-search-transcript-fixtures'
|
||||
import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache'
|
||||
function add(store: SessionSearchStore, id: number, cwd: string, text: string, count = 1) {
|
||||
store.db
|
||||
.prepare(
|
||||
`INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,message_count,resume_command) VALUES (?, 'claude', ?, ?, 'audit fixture', ?, ?, 1, '')`
|
||||
)
|
||||
.run(id, String(id), `/synthetic/${id}`, cwd, sessionSearchPathKey(cwd))
|
||||
for (let n = 0; n < count; n++) {
|
||||
const row = Number(
|
||||
store.db.prepare("INSERT INTO messages(session_row_id,role) VALUES (?, 'user')").run(id)
|
||||
.lastInsertRowid
|
||||
)
|
||||
store.db.prepare('INSERT INTO messages_fts(rowid,user_text) VALUES (?,?)').run(row, text)
|
||||
|
||||
/** The store keeps its connection private, so synthetic rows go in through a second one. */
|
||||
async function withIndex(
|
||||
run: (db: SyncDatabase, store: SessionSearchStore) => void
|
||||
): Promise<void> {
|
||||
const index = await openSessionSearchIndexFile('ss-query-regressions')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
try {
|
||||
run(index.db, store)
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
}
|
||||
|
||||
function add(
|
||||
db: SyncDatabase,
|
||||
id: number,
|
||||
cwd: string,
|
||||
text: string,
|
||||
count = 1,
|
||||
transcriptPath?: string
|
||||
) {
|
||||
db.prepare(
|
||||
`INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,message_count,resume_command) VALUES (?, 'claude', ?, ?, 'audit fixture', ?, ?, 1, '')`
|
||||
).run(id, String(id), `/synthetic/${id}`, cwd, sessionSearchPathKey(cwd, transcriptPath))
|
||||
for (let n = 0; n < count; n++) {
|
||||
const row = Number(
|
||||
db.prepare("INSERT INTO messages(session_row_id,role) VALUES (?, 'user')").run(id)
|
||||
.lastInsertRowid
|
||||
)
|
||||
db.prepare('INSERT INTO messages_fts(rowid,user_text) VALUES (?,?)').run(row, text)
|
||||
}
|
||||
}
|
||||
|
||||
describe('search correctness regressions', () => {
|
||||
it('finds a scoped match behind 600 out-of-scope rows', () => {
|
||||
const s = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(s, 1, '/unrelated', 'auditneedle', 600)
|
||||
add(s, 2, '/target', 'auditneedle padding')
|
||||
it('finds a scoped match behind 600 out-of-scope rows', async () => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, '/unrelated', 'auditneedle', 600)
|
||||
add(db, 2, '/target', 'auditneedle padding')
|
||||
expect(
|
||||
s.search({ query: 'auditneedle', scopePaths: ['/target'] }).hits.map((h) => h.sessionId)
|
||||
store.search({ query: 'auditneedle', scopePaths: ['/target'] }).hits.map((h) => h.sessionId)
|
||||
).toEqual(['2'])
|
||||
} finally {
|
||||
s.close()
|
||||
}
|
||||
})
|
||||
})
|
||||
it('falls back when the exact hit is out of scope', () => {
|
||||
const s = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(s, 1, '/unrelated', 'resolveTerminalPath')
|
||||
add(s, 2, '/target', 'resolve terminal path')
|
||||
|
||||
it('falls back when the exact hit is out of scope', async () => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, '/unrelated', 'resolveTerminalPath')
|
||||
add(db, 2, '/target', 'resolve terminal path')
|
||||
expect(
|
||||
s
|
||||
store
|
||||
.search({ query: 'resolveTerminalPath', scopePaths: ['/target'] })
|
||||
.hits.map((h) => h.sessionId)
|
||||
).toEqual(['2'])
|
||||
} finally {
|
||||
s.close()
|
||||
}
|
||||
})
|
||||
})
|
||||
it('phrase route requires adjacent ordered tokens', () => {
|
||||
const s = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(s, 1, '/target', 'beta separated alpha')
|
||||
expect(s.search({ query: '"alpha beta"' }).route).toBe('and')
|
||||
} finally {
|
||||
s.close()
|
||||
}
|
||||
|
||||
it('phrase route requires adjacent ordered tokens', async () => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, '/target', 'beta separated alpha')
|
||||
expect(store.search({ query: '"alpha beta"' }).route).toBe('and')
|
||||
})
|
||||
})
|
||||
it('unicode term indexed by FTS is searchable', () => {
|
||||
const s = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(s, 1, '/target', '안녕하세요')
|
||||
|
||||
it('unicode term indexed by FTS is searchable', async () => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, '/target', '안녕하세요')
|
||||
expect(
|
||||
s.db
|
||||
db
|
||||
.prepare('SELECT count(*) as n FROM messages_fts WHERE messages_fts MATCH ?')
|
||||
.get('안녕하세요')
|
||||
).toEqual({ n: 1 })
|
||||
expect(s.search({ query: '안녕하세요' }).hits).toHaveLength(1)
|
||||
} finally {
|
||||
s.close()
|
||||
}
|
||||
expect(store.search({ query: '안녕하세요' }).hits).toHaveLength(1)
|
||||
})
|
||||
})
|
||||
|
||||
it('ordinary list parse respects selected history retention', async () => {
|
||||
resetSessionParseCacheForTests()
|
||||
const root = await mkdtemp(join(tmpdir(), 'orca-audit-retention-'))
|
||||
@@ -93,7 +108,7 @@ describe('search correctness regressions', () => {
|
||||
await parseTranscript(path)
|
||||
expect(s.coverage().sessionsIndexed).toBe(0)
|
||||
} finally {
|
||||
s.dispose()
|
||||
await s.close()
|
||||
await removeTree(root)
|
||||
}
|
||||
})
|
||||
@@ -104,19 +119,25 @@ describe('path and retrieval contracts', () => {
|
||||
['C:\\Work\\App', 'c:/work/app', true],
|
||||
['C:\\Work\\App\\src', 'c:/work/app', true],
|
||||
['/work/APP/src', '/work/app', false],
|
||||
['/work/café', '/work/cafe\u0301', true],
|
||||
['/work/caf\u00e9', '/work/cafe\u0301', true],
|
||||
['/work/app-other', '/work/app', false],
|
||||
['/work/a_b/src', '/work/a_b', true],
|
||||
['/work/axb/src', '/work/a_b', false]
|
||||
])('scopes %s under %s: %s', (cwd, scope, expected) => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(store, 1, cwd, 'needle')
|
||||
expect(store.search({ query: 'needle', scopePaths: [scope] }).hits.length > 0).toBe(expected)
|
||||
['/work/axb/src', '/work/a_b', false],
|
||||
// Roots: both key to a degenerate prefix ('' and 'c:'), which is where a
|
||||
// range bound is easiest to get wrong. A Windows key is not under POSIX '/'.
|
||||
['/', '/', true],
|
||||
['/work/app', '/', true],
|
||||
['C:\\Work\\App', '/', false],
|
||||
['C:\\', 'C:\\', true],
|
||||
['C:\\Work\\App', 'C:\\', true]
|
||||
])('scopes %s under %s: %s', async (cwd, scope, expected) => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, cwd as string, 'needle')
|
||||
expect(store.search({ query: 'needle', scopePaths: [scope as string] }).hits.length > 0).toBe(
|
||||
expected
|
||||
)
|
||||
expect(store.search({ query: `needle path:"${scope}"` }).hits.length > 0).toBe(expected)
|
||||
} finally {
|
||||
store.close()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps WSL distro identity and Linux path case', () => {
|
||||
@@ -128,28 +149,53 @@ describe('path and retrieval contracts', () => {
|
||||
)
|
||||
})
|
||||
|
||||
it('newest returns distinct sessions even when one has over 600 matching rows', () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(store, 1, '/app', 'needle', 650)
|
||||
add(store, 2, '/app', 'needle padding')
|
||||
store.db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-06', 1)
|
||||
store.db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-05', 2)
|
||||
expect(
|
||||
store.search({ query: 'needle', sort: 'newest' }).hits.map((hit) => hit.sessionId)
|
||||
).toEqual(['1', '2'])
|
||||
} finally {
|
||||
store.close()
|
||||
// Both sorts collapse to one row per session before the candidate limit, so
|
||||
// the 650-row session cannot crowd the one-row session off the page.
|
||||
it.each(['relevance', 'newest'] as const)(
|
||||
'%s returns distinct sessions even when one has over 600 matching rows',
|
||||
async (sort) => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, '/app', 'needle', 650)
|
||||
add(db, 2, '/app', 'needle padding')
|
||||
db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-06', 1)
|
||||
db.prepare('UPDATE sessions SET updated_at = ? WHERE id = ?').run('2026-09-05', 2)
|
||||
expect(store.search({ query: 'needle', sort }).hits.map((hit) => hit.sessionId)).toEqual([
|
||||
'1',
|
||||
'2'
|
||||
])
|
||||
})
|
||||
}
|
||||
)
|
||||
|
||||
// The index writer can prove a WSL session's distro from its transcript path;
|
||||
// a query-time term never can. Orca stores a WSL workspace as the UNC path,
|
||||
// which is the spelling that keys the same way the writer did.
|
||||
it('scopes a WSL session by its UNC workspace path, not by the Linux spelling', async () => {
|
||||
await withIndex((db, store) => {
|
||||
add(
|
||||
db,
|
||||
1,
|
||||
'/home/ada/app',
|
||||
'needle',
|
||||
1,
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\ada\\session.jsonl'
|
||||
)
|
||||
const hits = (scope: string): number =>
|
||||
store.search({ query: 'needle', scopePaths: [scope] }).hits.length
|
||||
expect(hits('\\\\wsl$\\Ubuntu\\home\\ada\\app')).toBe(1)
|
||||
expect(hits('\\\\wsl.localhost\\Ubuntu\\home\\ada')).toBe(1)
|
||||
expect(hits('/home/ada/app')).toBe(0)
|
||||
expect(hits('\\\\wsl$\\Debian\\home\\ada\\app')).toBe(0)
|
||||
})
|
||||
})
|
||||
|
||||
it.each(['café', 'C', 'R', 'x', '修复'])('searches unicode61 token %s', (text) => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
try {
|
||||
add(store, 1, '/app', text)
|
||||
expect(store.search({ query: text }).hits).toHaveLength(1)
|
||||
} finally {
|
||||
store.close()
|
||||
it.each(['caf\u00e9', 'C', 'R', 'x', '\u4fee\u590d'])(
|
||||
'searches unicode61 token %s',
|
||||
async (text) => {
|
||||
await withIndex((db, store) => {
|
||||
add(db, 1, '/app', text)
|
||||
expect(store.search({ query: text }).hits).toHaveLength(1)
|
||||
})
|
||||
}
|
||||
})
|
||||
)
|
||||
})
|
||||
|
||||
@@ -5,12 +5,16 @@ import type {
|
||||
AiVaultSearchRoute
|
||||
} from '../../shared/ai-vault-search-types'
|
||||
import {
|
||||
AI_VAULT_SEARCH_LIMIT_DEFAULT,
|
||||
AI_VAULT_SEARCH_LIMIT_MAX,
|
||||
AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE,
|
||||
AI_VAULT_SEARCH_SNIPPET_MARK_OPEN
|
||||
} from '../../shared/ai-vault-search-types'
|
||||
import { sessionFields, type SessionRow } from './session-search-session-row'
|
||||
import {
|
||||
rankSessionHits,
|
||||
resolveLimit,
|
||||
sessionFields,
|
||||
type MessageRow,
|
||||
type SessionRow
|
||||
} from './session-search-hit-ranking'
|
||||
import {
|
||||
andExpression,
|
||||
orExpression,
|
||||
@@ -18,7 +22,7 @@ import {
|
||||
planSessionSearchQuery,
|
||||
type SessionSearchQueryPlan
|
||||
} from './session-search-query-planner'
|
||||
import { isCollapsibleContentHash } from './session-search-content-hash'
|
||||
import { VISIBLE_MESSAGES, VISIBLE_SESSIONS } from './session-search-schema'
|
||||
import { SessionSearchTypoRepair } from './session-search-typo-repair'
|
||||
import { sessionRowFilter, type SessionRowFilter } from './session-search-row-filter'
|
||||
import {
|
||||
@@ -31,29 +35,12 @@ const FULL_WEIGHTS = '3.0, 2.0, 1.0, 1.0'
|
||||
const CONVERSATION_WEIGHTS = '3.0, 2.0'
|
||||
// Candidate messages fetched before rolling up to sessions; more does not help.
|
||||
const MESSAGE_CANDIDATE_LIMIT = 600
|
||||
// Subtracted per session: `0.02 · ln(1 + messages)`; slightly positive on both eval sets.
|
||||
const LENGTH_PRIOR = 0.02
|
||||
const SNIPPET_TOKENS = 12
|
||||
// Why: single brackets are everywhere in code transcripts (`arr[0]`, regex
|
||||
// classes, markdown links) and would read as matches; doubled ones are rare.
|
||||
const SNIPPET_MARK_OPEN = AI_VAULT_SEARCH_SNIPPET_MARK_OPEN
|
||||
const SNIPPET_MARK_CLOSE = AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE
|
||||
|
||||
type MessageRow = {
|
||||
rowid: number
|
||||
score: number
|
||||
session_row_id: number
|
||||
role: string
|
||||
ts: string | null
|
||||
}
|
||||
|
||||
type ScoredSession = {
|
||||
session: SessionRow
|
||||
message: MessageRow
|
||||
score: number
|
||||
duplicateCount: number
|
||||
}
|
||||
|
||||
/** One search pass: the caller's args plus everything the operators decided. */
|
||||
type Retrieval = {
|
||||
args: AiVaultSearchArgs
|
||||
@@ -81,15 +68,9 @@ export class SessionSearchQuery {
|
||||
const retrieval: Retrieval = {
|
||||
args,
|
||||
tier: args.tier ?? 'full',
|
||||
filter: sessionRowFilter(args, split),
|
||||
filter: sessionRowFilter(args, split, cutoffMs),
|
||||
text: split.text
|
||||
}
|
||||
if (cutoffMs !== null) {
|
||||
retrieval.filter.conditions.push(
|
||||
'id IN (SELECT session_row_id FROM files WHERE mtime_ms >= ?)'
|
||||
)
|
||||
retrieval.filter.values.push(String(cutoffMs))
|
||||
}
|
||||
const plan = planSessionSearchQuery(retrieval.text)
|
||||
if (plan.terms.length === 0) {
|
||||
// Operators with no free text still name a scope, so answer with the
|
||||
@@ -124,7 +105,7 @@ export class SessionSearchQuery {
|
||||
return (
|
||||
this.db
|
||||
.prepare(
|
||||
`SELECT * FROM sessions ${where} ORDER BY updated_at DESC LIMIT ${resolveLimit(retrieval.args)}`
|
||||
`SELECT * FROM ${VISIBLE_SESSIONS} ${where} ORDER BY updated_at DESC LIMIT ${resolveLimit(retrieval.args)}`
|
||||
)
|
||||
.all(...values) as SessionRow[]
|
||||
).map((session) => ({
|
||||
@@ -172,22 +153,21 @@ export class SessionSearchQuery {
|
||||
private match(expression: string, retrieval: Retrieval): MessageRow[] {
|
||||
const { tier, filter, args } = retrieval
|
||||
const eligible = filter.conditions.length
|
||||
? ` AND m.session_row_id IN (SELECT id FROM sessions WHERE ${filter.conditions.join(' AND ')})`
|
||||
? ` AND m.session_row_id IN (SELECT id FROM ${VISIBLE_SESSIONS} WHERE ${filter.conditions.join(' AND ')})`
|
||||
: ''
|
||||
const table = tier === 'full' ? 'messages_fts' : 'conversation_fts'
|
||||
const weights = tier === 'full' ? FULL_WEIGHTS : CONVERSATION_WEIGHTS
|
||||
const matched = `SELECT ${table}.rowid AS rowid, -bm25(${table}, ${weights}) AS score,
|
||||
m.session_row_id, m.role, m.ts, s.updated_at
|
||||
FROM ${table} JOIN messages m ON m.id = ${table}.rowid
|
||||
JOIN sessions s ON s.id = m.session_row_id WHERE ${table} MATCH ?${eligible}
|
||||
AND (m.batch_id IS NULL OR m.batch_id NOT IN (SELECT id FROM search_write_batches WHERE published=0))`
|
||||
// Collapse messages before newest ordering so one long session cannot occupy the whole page.
|
||||
const sql =
|
||||
args.sort === 'newest'
|
||||
? `WITH matched AS MATERIALIZED (${matched})
|
||||
SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched
|
||||
GROUP BY session_row_id ORDER BY updated_at DESC, score DESC LIMIT ${MESSAGE_CANDIDATE_LIMIT}`
|
||||
: `${matched} ORDER BY score DESC LIMIT ${MESSAGE_CANDIDATE_LIMIT}`
|
||||
FROM ${table} JOIN ${VISIBLE_MESSAGES} m ON m.id = ${table}.rowid
|
||||
JOIN ${VISIBLE_SESSIONS} s ON s.id = m.session_row_id WHERE ${table} MATCH ?${eligible}`
|
||||
// Why: collapse to one row per session BEFORE the candidate limit, on both
|
||||
// sort orders, so a single long session cannot occupy the whole page.
|
||||
// `max(score)` makes SQLite pick that session's best row for the bare columns.
|
||||
const order = args.sort === 'newest' ? 'updated_at DESC, score DESC' : 'score DESC'
|
||||
const sql = `WITH matched AS MATERIALIZED (${matched})
|
||||
SELECT rowid, max(score) AS score, session_row_id, role, ts FROM matched
|
||||
GROUP BY session_row_id ORDER BY ${order} LIMIT ${MESSAGE_CANDIDATE_LIMIT}`
|
||||
return this.db.prepare(sql).all(expression, ...filter.values) as MessageRow[]
|
||||
}
|
||||
|
||||
@@ -196,46 +176,18 @@ export class SessionSearchQuery {
|
||||
retrieval: Retrieval,
|
||||
plan: SessionSearchQueryPlan
|
||||
): AiVaultSearchHit[] {
|
||||
const best = new Map<number, MessageRow>()
|
||||
for (const row of rows) {
|
||||
const current = best.get(row.session_row_id)
|
||||
if (!current || row.score > current.score) {
|
||||
best.set(row.session_row_id, row)
|
||||
}
|
||||
}
|
||||
// `match` already grouped to one best row per session.
|
||||
const best = new Map(rows.map((row) => [row.session_row_id, row]))
|
||||
if (best.size === 0) {
|
||||
return []
|
||||
}
|
||||
const sessions = this.loadSessions([...best.keys()], retrieval.filter)
|
||||
const scored = collapseForks(
|
||||
sessions.map((session) => {
|
||||
const message = best.get(session.id)!
|
||||
return {
|
||||
session,
|
||||
message,
|
||||
score: message.score - LENGTH_PRIOR * Math.log(1 + session.message_count),
|
||||
duplicateCount: 1
|
||||
}
|
||||
})
|
||||
)
|
||||
scored.sort((left, right) =>
|
||||
retrieval.args.sort === 'newest'
|
||||
? (right.session.updated_at ?? '').localeCompare(left.session.updated_at ?? '')
|
||||
: right.score - left.score
|
||||
)
|
||||
const table = retrieval.tier === 'full' ? 'messages_fts' : 'conversation_fts'
|
||||
return scored
|
||||
.slice(0, resolveLimit(retrieval.args))
|
||||
.map(({ session, message, score, duplicateCount }) => ({
|
||||
...sessionFields(session),
|
||||
score,
|
||||
...(duplicateCount > 1 ? { duplicateCount } : {}),
|
||||
evidence: {
|
||||
role: message.role as AiVaultSearchHit['evidence']['role'],
|
||||
timestamp: message.ts,
|
||||
snippet: this.snippet(table, message.rowid, plan)
|
||||
}
|
||||
}))
|
||||
return rankSessionHits(
|
||||
this.loadSessions([...best.keys()], retrieval.filter),
|
||||
best,
|
||||
retrieval.args,
|
||||
(message) => this.snippet(table, message.rowid, plan)
|
||||
)
|
||||
}
|
||||
|
||||
// Why: the snippet must highlight the terms that actually retrieved the row,
|
||||
@@ -274,50 +226,7 @@ export class SessionSearchQuery {
|
||||
private loadSessions(ids: number[], filter: SessionRowFilter): SessionRow[] {
|
||||
const conditions = [`id IN (${ids.map(() => '?').join(',')})`, ...filter.conditions]
|
||||
return this.db
|
||||
.prepare(`SELECT * FROM sessions WHERE ${conditions.join(' AND ')}`)
|
||||
.prepare(`SELECT * FROM ${VISIBLE_SESSIONS} WHERE ${conditions.join(' AND ')}`)
|
||||
.all(...ids, ...filter.values) as SessionRow[]
|
||||
}
|
||||
}
|
||||
|
||||
// Why: the desktop IPC forwards its payload unvalidated, so a non-positive
|
||||
// limit must be clamped here or `LIMIT -1` / `slice(0, -1)` leak through.
|
||||
function resolveLimit(args: AiVaultSearchArgs): number {
|
||||
const requested = Number.isInteger(args.limit)
|
||||
? (args.limit as number)
|
||||
: AI_VAULT_SEARCH_LIMIT_DEFAULT
|
||||
return Math.min(Math.max(1, requested), AI_VAULT_SEARCH_LIMIT_MAX)
|
||||
}
|
||||
|
||||
/**
|
||||
* Folds forked copies of one conversation into a single hit: same opening
|
||||
* prefix, newest `updated_at` wins, the rest become `duplicateCount`. Done here
|
||||
* and not at write time so index rows stay per file (cursors and deletes).
|
||||
*/
|
||||
function collapseForks(scored: ScoredSession[]): ScoredSession[] {
|
||||
const groups = new Map<string, ScoredSession[]>()
|
||||
for (const entry of scored) {
|
||||
const { content_hash: hash, content_hash_count: count, id } = entry.session
|
||||
const key = isCollapsibleContentHash(hash, count) ? `hash:${hash}` : `session:${id}`
|
||||
const group = groups.get(key)
|
||||
if (group) {
|
||||
group.push(entry)
|
||||
} else {
|
||||
groups.set(key, [entry])
|
||||
}
|
||||
}
|
||||
const collapsed: ScoredSession[] = []
|
||||
for (const group of groups.values()) {
|
||||
if (group.length === 1) {
|
||||
collapsed.push(group[0]!)
|
||||
continue
|
||||
}
|
||||
const winner = group.reduce((best, entry) => (isNewer(entry, best) ? entry : best))
|
||||
collapsed.push({ ...winner, duplicateCount: group.length })
|
||||
}
|
||||
return collapsed
|
||||
}
|
||||
|
||||
function isNewer(entry: ScoredSession, best: ScoredSession): boolean {
|
||||
const order = (entry.session.updated_at ?? '').localeCompare(best.session.updated_at ?? '')
|
||||
return order === 0 ? entry.score > best.score : order > 0
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ import { sessionCandidatesFromDiscoveries } from '../ai-vault/session-scanner-ca
|
||||
import type { SessionFileCandidate } from '../ai-vault/session-scanner-types'
|
||||
|
||||
import { waitForPromiseWithSignal, throwIfSignalAborted } from '../../shared/abort-signal-reason'
|
||||
import { stableInFlightKey } from '../../shared/in-flight-promise-dedupe'
|
||||
import type { SessionSearchScanRoots } from './session-search-service'
|
||||
|
||||
type Refresh = { controller: AbortController; promise: Promise<void>; users: number }
|
||||
@@ -18,7 +19,7 @@ export class SessionSearchRefreshLane {
|
||||
signal?: AbortSignal
|
||||
): Promise<void> {
|
||||
throwIfSignalAborted(signal)
|
||||
const key = JSON.stringify(Object.entries(roots).sort(([a], [b]) => a.localeCompare(b)))
|
||||
const key = stableInFlightKey(Object.entries(roots).sort(([a], [b]) => a.localeCompare(b)))
|
||||
let run = this.runs.get(key)
|
||||
if (!run) {
|
||||
const controller = new AbortController()
|
||||
|
||||
@@ -1,16 +1,13 @@
|
||||
import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import { mkdtemp } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { expect, it } from 'vitest'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { openSessionSearchIndexFile } from './session-search-staged-write-test-fixture'
|
||||
import {
|
||||
deleteExpiredSearchFiles,
|
||||
RETENTION_DELETE_ROWS_PER_STEP
|
||||
} from './session-search-retention-delete'
|
||||
|
||||
function seed(store: SessionSearchStore, id: number, rows: number, mtime: number) {
|
||||
const db = store.db
|
||||
function seed(db: SyncDatabase, id: number, rows: number, mtime: number) {
|
||||
db.prepare(`INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,resume_command)
|
||||
VALUES (?, 'claude', ?, ?, 'synthetic retention', '/fixture', '/fixture', '')`).run(
|
||||
id,
|
||||
@@ -37,21 +34,22 @@ function seed(store: SessionSearchStore, id: number, rows: number, mtime: number
|
||||
}
|
||||
|
||||
it('yields within a large file while hiding partial rows and preserving unrelated sessions', async () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
seed(store, 1, 1025, 1)
|
||||
seed(store, 2, 1, 200)
|
||||
const index = await openSessionSearchIndexFile('ss-retention-yield')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
seed(index.db, 1, 1025, 1)
|
||||
seed(index.db, 2, 1, 200)
|
||||
let previous = 1025
|
||||
let steps = 0
|
||||
try {
|
||||
await deleteExpiredSearchFiles(
|
||||
store.db,
|
||||
index.db,
|
||||
100,
|
||||
() => false,
|
||||
() => {},
|
||||
async () => {
|
||||
const left = Number(
|
||||
(
|
||||
store.db.prepare('SELECT count(*) AS n FROM messages WHERE session_row_id=1').get() as {
|
||||
index.db.prepare('SELECT count(*) AS n FROM messages WHERE session_row_id=1').get() as {
|
||||
n: number
|
||||
}
|
||||
).n
|
||||
@@ -66,25 +64,25 @@ it('yields within a large file while hiding partial rows and preserving unrelate
|
||||
}
|
||||
)
|
||||
expect(steps).toBe(5)
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 1 })
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM conversation_fts').get()).toEqual({ n: 1 })
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM search_pending_deletes').get()).toEqual({
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 1 })
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM conversation_fts').get()).toEqual({ n: 1 })
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM search_pending_deletes').get()).toEqual({
|
||||
n: 0
|
||||
})
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('finishes an interrupted deletion after reopening even when history becomes unlimited', async () => {
|
||||
const root = await mkdtemp(join(tmpdir(), 'ss-retention-resume-'))
|
||||
const path = join(root, 'index.sqlite')
|
||||
let store = new SessionSearchStore(path)
|
||||
const index = await openSessionSearchIndexFile('ss-retention-resume')
|
||||
let store = new SessionSearchStore(index.path)
|
||||
let closed = false
|
||||
try {
|
||||
seed(store, 1, 513, 1)
|
||||
seed(index.db, 1, 513, 1)
|
||||
await deleteExpiredSearchFiles(
|
||||
store.db,
|
||||
index.db,
|
||||
100,
|
||||
() => closed,
|
||||
() => {},
|
||||
@@ -93,30 +91,31 @@ it('finishes an interrupted deletion after reopening even when history becomes u
|
||||
closed = true
|
||||
}
|
||||
)
|
||||
store = new SessionSearchStore(path)
|
||||
store = new SessionSearchStore(index.path)
|
||||
closed = false
|
||||
expect(store.search({ query: 'retentionneedle' }).hits).toEqual([])
|
||||
expect(store.coverage().sessionsIndexed).toBe(0)
|
||||
await store.purgeOlderThan(null)
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 0 })
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ n: 0 })
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 0 })
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ n: 0 })
|
||||
} finally {
|
||||
if (!closed) {
|
||||
store.close()
|
||||
}
|
||||
await removeTree(root)
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('does not orphan a replacement file when resuming an older deletion for the same path', async () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const index = await openSessionSearchIndexFile('ss-retention-reused-path')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
try {
|
||||
seed(store, 1, 2, 1)
|
||||
store.db.exec(
|
||||
seed(index.db, 1, 2, 1)
|
||||
index.db.exec(
|
||||
"INSERT INTO search_pending_deletes(path,session_row_id) VALUES ('1',1); DELETE FROM files WHERE path='1'"
|
||||
)
|
||||
seed(store, 2, 2, 1)
|
||||
store.db.exec("UPDATE files SET path='1' WHERE path='2'")
|
||||
seed(index.db, 2, 2, 1)
|
||||
index.db.exec("UPDATE files SET path='1' WHERE path='2'")
|
||||
await store.purgeOlderThan(100)
|
||||
for (const table of [
|
||||
'messages',
|
||||
@@ -126,28 +125,31 @@ it('does not orphan a replacement file when resuming an older deletion for the s
|
||||
'files',
|
||||
'search_pending_deletes'
|
||||
]) {
|
||||
expect(store.db.prepare(`SELECT count(*) AS n FROM ${table}`).get()).toEqual({ n: 0 })
|
||||
expect(index.db.prepare(`SELECT count(*) AS n FROM ${table}`).get()).toEqual({ n: 0 })
|
||||
}
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('cancels retention between batches and resumes without exposing a partial session', async () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const index = await openSessionSearchIndexFile('ss-retention-cancel')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
try {
|
||||
seed(store, 1, 1025, 1)
|
||||
seed(index.db, 1, 1025, 1)
|
||||
const controller = new AbortController()
|
||||
const purge = store.purgeOlderThan(100, controller.signal)
|
||||
setImmediate(() => controller.abort())
|
||||
await purge
|
||||
const remaining = store.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }
|
||||
const remaining = index.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }
|
||||
expect(remaining.n).toBeGreaterThan(0)
|
||||
expect(remaining.n).toBeLessThan(1025)
|
||||
expect(store.search({ query: 'retentionneedle' }).hits).toEqual([])
|
||||
await store.purgeOlderThan(null)
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 0 })
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 0 })
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
import { assertSearchWalBudget } from './session-search-wal-budget'
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { inSessionParseFileLane } from '../ai-vault/session-parse-file-lane'
|
||||
@@ -55,7 +54,6 @@ export async function deleteExpiredSearchFiles(
|
||||
if (!pending) {
|
||||
return
|
||||
}
|
||||
assertSearchWalBudget(db)
|
||||
db.exec('BEGIN IMMEDIATE')
|
||||
try {
|
||||
const ids = db
|
||||
|
||||
@@ -1,12 +1,15 @@
|
||||
import { sessionSearchPathKey } from './session-search-path-key'
|
||||
import type { AiVaultSearchArgs } from '../../shared/ai-vault-search-types'
|
||||
import type { AiVaultSearchQuerySplit } from '../../shared/ai-vault-search-query-operators'
|
||||
import { isRuntimePathAbsolute } from '../../shared/cross-platform-path'
|
||||
import {
|
||||
isRuntimePathAbsolute,
|
||||
normalizeRuntimePathSeparators
|
||||
} from '../../shared/cross-platform-path'
|
||||
|
||||
/** SQL fragments for the `sessions` WHERE clause; every condition is ANDed. */
|
||||
export type SessionRowFilter = {
|
||||
conditions: string[]
|
||||
values: string[]
|
||||
values: (string | number)[]
|
||||
}
|
||||
|
||||
// Stored identity preserves execution-host case and WSL distro semantics.
|
||||
@@ -15,16 +18,28 @@ const CWD = 'cwd_key'
|
||||
// last segment off, leaving the parent prefix to delete out of p.
|
||||
const CWD_BASENAME = `replace(${CWD}, rtrim(${CWD}, replace(${CWD}, '/', '')), '')`
|
||||
|
||||
/**
|
||||
* Every caller-supplied narrowing in one place, so `match`, `recent`, and
|
||||
* `loadSessions` cannot drift apart. Row visibility is not here: it belongs to
|
||||
* the `visible_sessions` / `visible_messages` views these conditions run over.
|
||||
*
|
||||
* Case rule, one for the whole file: a comparison that claims *identity*
|
||||
* (`scopePaths`, an absolute `path:`) compares the stored key as-is, so it folds
|
||||
* exactly where the execution host folds — Windows drives and the WSL distro
|
||||
* segment, never a POSIX directory name. A comparison that is only a *substring
|
||||
* probe* (a relative `path:`, any `repo:`) uses LIKE, which folds ASCII and
|
||||
* nothing else; SQLite has no Unicode fold, and `lower()` would fold ASCII twice
|
||||
* while still missing `É`, so it is not used.
|
||||
*/
|
||||
export function sessionRowFilter(
|
||||
args: AiVaultSearchArgs,
|
||||
split: AiVaultSearchQuerySplit
|
||||
split: AiVaultSearchQuerySplit,
|
||||
cutoffMs: number | null = null
|
||||
): SessionRowFilter {
|
||||
const filter: SessionRowFilter = {
|
||||
conditions: [
|
||||
'index_ready = 1',
|
||||
'id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL)'
|
||||
],
|
||||
values: []
|
||||
const filter: SessionRowFilter = { conditions: [], values: [] }
|
||||
if (cutoffMs !== null) {
|
||||
filter.conditions.push('id IN (SELECT session_row_id FROM files WHERE mtime_ms >= ?)')
|
||||
filter.values.push(cutoffMs)
|
||||
}
|
||||
if (args.agents && args.agents.length > 0) {
|
||||
filter.conditions.push(`agent IN (${args.agents.map(() => '?').join(',')})`)
|
||||
@@ -35,53 +50,65 @@ export function sessionRowFilter(
|
||||
filter.values.push(args.since)
|
||||
}
|
||||
if (args.scopePaths && args.scopePaths.length > 0) {
|
||||
filter.conditions.push(
|
||||
`(${args.scopePaths.map(() => `(${CWD} = ? OR substr(${CWD}, 1, length(?)) = ?)`).join(' OR ')})`
|
||||
addGroup(
|
||||
filter,
|
||||
args.scopePaths.map((scope) => insideCondition(filter, sessionSearchPathKey(scope)))
|
||||
)
|
||||
for (const scope of args.scopePaths) {
|
||||
// Literal prefixes keep `%` and `_` in folder names from widening scope.
|
||||
const normalized = sessionSearchPathKey(scope)
|
||||
filter.values.push(normalized, `${normalized}/`, `${normalized}/`)
|
||||
}
|
||||
}
|
||||
// Operators narrow, never widen: each one is its own ANDed condition on top
|
||||
// of whatever scope the caller already asked for.
|
||||
for (const term of split.pathTerms) {
|
||||
addPathTerm(filter, term)
|
||||
}
|
||||
for (const term of split.repoTerms) {
|
||||
// Operators narrow the caller's scope, never widen it, and follow the usual
|
||||
// qualifier semantics: OR within one key, AND across keys, so `path:a path:b`
|
||||
// means either while `repo:x path:a` means both.
|
||||
addGroup(
|
||||
filter,
|
||||
split.pathTerms.map((term) => pathTermCondition(filter, term))
|
||||
)
|
||||
addGroup(
|
||||
filter,
|
||||
// Why: a folder workspace has no repo name beyond its own folder, so the
|
||||
// last segment of cwd is the only honest local proxy for `repo:`.
|
||||
filter.conditions.push(`lower(${CWD_BASENAME}) LIKE ? ESCAPE '\\'`)
|
||||
filter.values.push(likeContains(term))
|
||||
}
|
||||
split.repoTerms.map((term) => containsCondition(filter, CWD_BASENAME, term))
|
||||
)
|
||||
return filter
|
||||
}
|
||||
|
||||
function addPathTerm(filter: SessionRowFilter, term: string): void {
|
||||
const normalized = normalizeCwdTerm(term)
|
||||
if (!normalized) {
|
||||
return
|
||||
function addGroup(filter: SessionRowFilter, conditions: (string | null)[]): void {
|
||||
const present = conditions.filter((condition) => condition !== null)
|
||||
if (present.length > 0) {
|
||||
filter.conditions.push(`(${present.join(' OR ')})`)
|
||||
}
|
||||
}
|
||||
|
||||
function pathTermCondition(filter: SessionRowFilter, term: string): string | null {
|
||||
if (isRuntimePathAbsolute(term)) {
|
||||
const key = sessionSearchPathKey(term)
|
||||
filter.conditions.push(`(${CWD} = ? OR substr(${CWD}, 1, length(?)) = ?)`)
|
||||
filter.values.push(key, `${key}/`, `${key}/`)
|
||||
return
|
||||
return insideCondition(filter, sessionSearchPathKey(term))
|
||||
}
|
||||
filter.conditions.push(`lower(${CWD}) LIKE ? ESCAPE '\\'`)
|
||||
filter.values.push(likeContains(normalized))
|
||||
// A bare fragment cannot prove Windows semantics, so fold separators anyway:
|
||||
// `path:Work\App` is a Windows user typing, never a POSIX file named `Work\App`.
|
||||
const fragment = normalizeRuntimePathSeparators(term.normalize('NFC')).replace(/\/+$/, '')
|
||||
return fragment ? containsCondition(filter, CWD, fragment) : null
|
||||
}
|
||||
|
||||
function normalizeCwdTerm(term: string): string {
|
||||
return term.replaceAll('\\', '/').replace(/\/+$/, '').toLowerCase()
|
||||
/**
|
||||
* `key` itself, or anything below it. Why a half-open range and not
|
||||
* `substr(key, 1, length(?)) = ?`: only `>=`/`<` can seek `sessions_cwd_key`;
|
||||
* the substr form scans it. The bound is the child prefix with its last byte
|
||||
* incremented, so it stops at the end of that prefix and nowhere else. The two
|
||||
* arms cannot merge: one range over the bare key would also swallow a sibling
|
||||
* like `/work/app-other`. No wildcards, so `%`/`_` in a folder name are literal.
|
||||
*/
|
||||
function insideCondition(filter: SessionRowFilter, key: string): string {
|
||||
const children = `${key}/`
|
||||
filter.values.push(key, children, nextAfterPrefix(children))
|
||||
return `(${CWD} = ? OR (${CWD} >= ? AND ${CWD} < ?))`
|
||||
}
|
||||
|
||||
function likeContains(value: string): string {
|
||||
return `%${escapeLike(value)}%`
|
||||
/** The first string that sorts after every string starting with `prefix`. */
|
||||
function nextAfterPrefix(prefix: string): string {
|
||||
return prefix.slice(0, -1) + String.fromCharCode(prefix.charCodeAt(prefix.length - 1) + 1)
|
||||
}
|
||||
|
||||
// LIKE wildcards inside a user-typed term are literal text, not a pattern.
|
||||
function escapeLike(value: string): string {
|
||||
return value.replaceAll(/[\\%_]/g, '\\$&')
|
||||
function containsCondition(filter: SessionRowFilter, column: string, term: string): string {
|
||||
// LIKE wildcards inside a user-typed term are literal text, not a pattern.
|
||||
filter.values.push(`%${term.normalize('NFC').replaceAll(/[\\%_]/g, '\\$&')}%`)
|
||||
return `${column} LIKE ? ESCAPE '\\'`
|
||||
}
|
||||
|
||||
@@ -1,15 +1,34 @@
|
||||
import type * as NodeFs from 'node:fs'
|
||||
import { mkdtemp, stat, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import {
|
||||
removeTree,
|
||||
WINDOWS_RM_MAX_RETRIES,
|
||||
WINDOWS_RM_RETRY_DELAY_MS
|
||||
} from '../../shared/windows-transient-lock-removal'
|
||||
import SyncDatabase from '../sqlite/sync-database'
|
||||
import {
|
||||
SESSION_SEARCH_SCHEMA_VERSION,
|
||||
openSessionSearchDatabase,
|
||||
removeSessionSearchDatabase
|
||||
removeSessionSearchDatabase,
|
||||
VISIBLE_MESSAGES,
|
||||
VISIBLE_SESSIONS
|
||||
} from './session-search-schema'
|
||||
|
||||
const recordedRmSync = vi.hoisted(() => vi.fn())
|
||||
vi.mock('node:fs', async () => {
|
||||
const actual = await vi.importActual<typeof NodeFs>('node:fs')
|
||||
return {
|
||||
...actual,
|
||||
rmSync: (...args: Parameters<typeof actual.rmSync>) => {
|
||||
recordedRmSync(...args)
|
||||
return actual.rmSync(...args)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
let roots: string[] = []
|
||||
|
||||
afterEach(async () => {
|
||||
@@ -99,3 +118,48 @@ it('closes the SQLite handle when corrupt data fails initialization', async () =
|
||||
const recovered = openSessionSearchDatabase(path)
|
||||
recovered.close()
|
||||
})
|
||||
|
||||
describe('visibility views', () => {
|
||||
it('hides a staging session, a tombstoned session and an in-flight batch', async () => {
|
||||
const db = openSessionSearchDatabase(await tempDatabasePath())
|
||||
try {
|
||||
db.exec(`INSERT INTO sessions(id,index_ready,agent,session_id,file_path,title,resume_command)
|
||||
VALUES (1,1,'claude','a','a','published',''),(2,0,'claude','b','b','staging',''),
|
||||
(3,1,'claude','c','c','tombstoned','');
|
||||
INSERT INTO search_pending_deletes(path,session_row_id) VALUES ('c',3);
|
||||
INSERT INTO search_write_batches(id,session_row_id) VALUES (7,1);
|
||||
INSERT INTO messages(id,session_row_id,batch_id,role) VALUES (1,1,NULL,'user'),(2,1,7,'user')`)
|
||||
expect(db.prepare(`SELECT title FROM ${VISIBLE_SESSIONS} ORDER BY id`).all()).toEqual([
|
||||
{ title: 'published' }
|
||||
])
|
||||
expect(db.prepare(`SELECT id FROM ${VISIBLE_MESSAGES} ORDER BY id`).all()).toEqual([
|
||||
{ id: 1 }
|
||||
])
|
||||
// Publish clears the pointer, so visibility never depends on the batch row surviving.
|
||||
db.exec(
|
||||
'UPDATE messages SET batch_id=NULL WHERE batch_id=7; DELETE FROM search_write_batches'
|
||||
)
|
||||
expect(db.prepare(`SELECT count(*) AS n FROM ${VISIBLE_MESSAGES}`).get()).toEqual({ n: 2 })
|
||||
} finally {
|
||||
db.close()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
it('gives Windows the shared retry options for a late handle release', async () => {
|
||||
const path = await tempDatabasePath()
|
||||
vi.spyOn(process, 'platform', 'get').mockReturnValue('win32')
|
||||
recordedRmSync.mockClear()
|
||||
try {
|
||||
removeSessionSearchDatabase(path)
|
||||
expect(recordedRmSync).toHaveBeenCalled()
|
||||
for (const [, options] of recordedRmSync.mock.calls) {
|
||||
expect(options).toMatchObject({
|
||||
maxRetries: WINDOWS_RM_MAX_RETRIES,
|
||||
retryDelay: WINDOWS_RM_RETRY_DELAY_MS
|
||||
})
|
||||
}
|
||||
} finally {
|
||||
vi.restoreAllMocks()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
import { rmSync } from 'node:fs'
|
||||
import SyncDatabase from '../sqlite/sync-database'
|
||||
import { transientLockRemovalOptions } from '../../shared/windows-transient-lock-removal'
|
||||
|
||||
// Bump to drop and rebuild: the index is a cache over the transcripts, never a source.
|
||||
export const SESSION_SEARCH_SCHEMA_VERSION = 9
|
||||
export const SESSION_SEARCH_SCHEMA_VERSION = 10
|
||||
|
||||
// unicode61 keeps `_ . - /` inside tokens so paths and identifiers match exactly;
|
||||
// the `identifiers` column carries the split form (see session-search-identifier-split).
|
||||
@@ -10,6 +11,10 @@ export const SESSION_SEARCH_SCHEMA_VERSION = 9
|
||||
// left out so `#123` still answers a search for `123`.
|
||||
const TOKENIZER = `tokenize="unicode61 tokenchars '_.-/+'"`
|
||||
|
||||
/** Sessions and messages a read may return: published, not tombstoned. */
|
||||
export const VISIBLE_SESSIONS = 'visible_sessions'
|
||||
export const VISIBLE_MESSAGES = 'visible_messages'
|
||||
|
||||
const SCHEMA_SQL = `
|
||||
CREATE TABLE IF NOT EXISTS meta(key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
CREATE TABLE IF NOT EXISTS sessions(
|
||||
@@ -35,6 +40,7 @@ CREATE TABLE IF NOT EXISTS sessions(
|
||||
CREATE INDEX IF NOT EXISTS sessions_agent ON sessions(agent);
|
||||
CREATE INDEX IF NOT EXISTS sessions_content_hash ON sessions(content_hash);
|
||||
CREATE INDEX IF NOT EXISTS sessions_updated_at ON sessions(updated_at);
|
||||
CREATE INDEX IF NOT EXISTS sessions_cwd_key ON sessions(cwd_key);
|
||||
CREATE TABLE IF NOT EXISTS files(
|
||||
path TEXT PRIMARY KEY,
|
||||
dev INTEGER,
|
||||
@@ -49,13 +55,12 @@ CREATE TABLE IF NOT EXISTS search_pending_deletes(
|
||||
session_row_id INTEGER NOT NULL,
|
||||
batch_id INTEGER
|
||||
);
|
||||
-- A row exists only while its batch is in flight; publish clears its messages and deletes it.
|
||||
CREATE TABLE IF NOT EXISTS search_write_batches(
|
||||
id INTEGER PRIMARY KEY,
|
||||
session_row_id INTEGER NOT NULL,
|
||||
published INTEGER NOT NULL DEFAULT 0
|
||||
session_row_id INTEGER NOT NULL
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS search_write_batches_session ON search_write_batches(session_row_id);
|
||||
CREATE INDEX IF NOT EXISTS search_write_batches_pending ON search_write_batches(published);
|
||||
CREATE TABLE IF NOT EXISTS messages(
|
||||
id INTEGER PRIMARY KEY,
|
||||
session_row_id INTEGER NOT NULL,
|
||||
@@ -80,6 +85,13 @@ CREATE TABLE IF NOT EXISTS search_log(
|
||||
hits INTEGER NOT NULL,
|
||||
duration_ms REAL NOT NULL
|
||||
);
|
||||
-- Why: staged rows must never reach a result. One definition per half, so a new
|
||||
-- read site cannot forget one; SQLite flattens both into the caller's plan.
|
||||
CREATE VIEW IF NOT EXISTS ${VISIBLE_SESSIONS} AS SELECT * FROM sessions
|
||||
WHERE index_ready = 1
|
||||
AND id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL);
|
||||
CREATE VIEW IF NOT EXISTS ${VISIBLE_MESSAGES} AS SELECT * FROM messages
|
||||
WHERE batch_id IS NULL;
|
||||
`
|
||||
|
||||
export function openSessionSearchDatabase(path: string): SyncDatabase {
|
||||
@@ -122,16 +134,10 @@ export function removeSessionSearchDatabase(path: string): void {
|
||||
return
|
||||
}
|
||||
for (const suffix of ['', '-wal', '-shm', '-journal']) {
|
||||
rmSync(`${path}${suffix}`, { force: true })
|
||||
rmSync(`${path}${suffix}`, transientLockRemovalOptions())
|
||||
}
|
||||
}
|
||||
|
||||
export function openSessionSearchDatabaseReadOnly(path: string): SyncDatabase {
|
||||
const db = new SyncDatabase(path, { readonly: true, fileMustExist: true })
|
||||
db.pragma('busy_timeout = 1500')
|
||||
return db
|
||||
}
|
||||
|
||||
function readSchemaVersion(db: SyncDatabase): number | null {
|
||||
const table = db
|
||||
.prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'meta'")
|
||||
|
||||
@@ -8,7 +8,10 @@ import type {
|
||||
AiVaultSearchCoverage,
|
||||
AiVaultSearchResult
|
||||
} from '../../shared/ai-vault-search-types'
|
||||
import { DISABLED_AI_VAULT_SEARCH_COVERAGE as DISABLED_COVERAGE } from '../../shared/ai-vault-search-coverage'
|
||||
import {
|
||||
DISABLED_AI_VAULT_SEARCH_COVERAGE as DISABLED_COVERAGE,
|
||||
NO_AI_VAULT_SEARCH_INDEX_RESULT as NO_INDEX_RESULT
|
||||
} from '../../shared/ai-vault-search-coverage'
|
||||
import {
|
||||
aiVaultSearchHistoryCutoffMs,
|
||||
narrowsAiVaultSearchHistory,
|
||||
@@ -20,7 +23,10 @@ import { sessionCandidatesFromDiscoveries } from '../ai-vault/session-scanner-ca
|
||||
import { discoverAiVaultSessionSources } from '../ai-vault/session-scanner-source-discovery'
|
||||
import type { AiVaultScanIssue } from '../../shared/ai-vault-types'
|
||||
import type { AiVaultScanOptions, SessionFileCandidate } from '../ai-vault/session-scanner-types'
|
||||
import { registerSessionSearchIndexSink } from '../ai-vault/session-search-capture'
|
||||
import {
|
||||
getSessionSearchIndexSink,
|
||||
registerSessionSearchIndexSink
|
||||
} from '../ai-vault/session-search-capture'
|
||||
import { parseSearchCandidates } from './session-search-parse-candidates'
|
||||
import { removeSessionSearchDatabase } from './session-search-schema'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
@@ -53,7 +59,7 @@ export class SessionSearchService {
|
||||
this.databasePath = options.databasePath
|
||||
this.policy = { ...options }
|
||||
if (this.policy.enabled) {
|
||||
this.open()
|
||||
this.applyPolicyToStore(this.openStore())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,9 +69,8 @@ export class SessionSearchService {
|
||||
roots: SessionSearchScanRoots,
|
||||
signal?: AbortSignal
|
||||
): Promise<AiVaultSearchResult> {
|
||||
const store = this.store
|
||||
if (!store) {
|
||||
return { hits: [], route: 'and', durationMs: 0, coverage: DISABLED_COVERAGE }
|
||||
if (!this.store) {
|
||||
return NO_INDEX_RESULT
|
||||
}
|
||||
const backfill = this.ensureBackfill(roots)
|
||||
// Why: the backfill parses in this same process and an 80 MB transcript
|
||||
@@ -80,7 +85,7 @@ export class SessionSearchService {
|
||||
async (sharedSignal) => {
|
||||
await this.parseAll(
|
||||
this.withinHistory(await discoverRecentSearchFiles(roots, sharedSignal)),
|
||||
sharedSignal
|
||||
{ signal: sharedSignal }
|
||||
)
|
||||
await this.reindexStale(sharedSignal)
|
||||
},
|
||||
@@ -88,9 +93,12 @@ export class SessionSearchService {
|
||||
)
|
||||
}
|
||||
void backfill
|
||||
// Why: presence checks await between query rounds, and a configure() in
|
||||
// that window closes the database; read the handle per round so a closed
|
||||
// store yields no hits instead of throwing on a freed statement.
|
||||
return await searchPresentSessionSources(
|
||||
args,
|
||||
(query) => store.search(query),
|
||||
(query) => this.store?.search(query) ?? NO_INDEX_RESULT,
|
||||
(paths) => this.invalidate(paths),
|
||||
signal
|
||||
)
|
||||
@@ -138,8 +146,6 @@ export class SessionSearchService {
|
||||
const wasEnabled = this.policy.enabled
|
||||
const previousDays = this.policy.historyDays
|
||||
this.policy = { ...next }
|
||||
this.store?.indexing.setPaused(next.paused === true)
|
||||
this.store?.setHistoryDays(next.historyDays)
|
||||
if (options.clearIndex || (wasEnabled && !next.enabled)) {
|
||||
await this.stop()
|
||||
} else if (wasEnabled && (next.paused || previousDays !== next.historyDays)) {
|
||||
@@ -152,9 +158,8 @@ export class SessionSearchService {
|
||||
if (!next.enabled) {
|
||||
return DISABLED_COVERAGE
|
||||
}
|
||||
const store = this.store ?? this.open()
|
||||
store.setAcceptingWrites(!next.paused)
|
||||
store.indexing.setPaused(next.paused === true)
|
||||
const store = this.store ?? this.openStore()
|
||||
this.applyPolicyToStore(store)
|
||||
const cutoff = aiVaultSearchHistoryCutoffMs(next.historyDays)
|
||||
if (
|
||||
wasEnabled &&
|
||||
@@ -208,23 +213,24 @@ export class SessionSearchService {
|
||||
}
|
||||
}
|
||||
|
||||
dispose(): void {
|
||||
this.backfillController?.abort()
|
||||
this.closeStore()
|
||||
}
|
||||
|
||||
async close(): Promise<void> {
|
||||
await this.stop({ drainRefreshes: true })
|
||||
}
|
||||
|
||||
private open(): SessionSearchStore {
|
||||
/** Creation only; the caller applies the policy before anything can await. */
|
||||
private openStore(): SessionSearchStore {
|
||||
mkdirSync(dirname(this.databasePath), { recursive: true })
|
||||
this.store = new SessionSearchStore(this.databasePath)
|
||||
this.store.setHistoryDays(this.policy.historyDays)
|
||||
this.store.setAcceptingWrites(!this.policy.paused)
|
||||
this.store.indexing.setPaused(this.policy.paused === true)
|
||||
registerSessionSearchIndexSink(this.store)
|
||||
return this.store
|
||||
const store = new SessionSearchStore(this.databasePath)
|
||||
this.store = store
|
||||
registerSessionSearchIndexSink(store)
|
||||
return store
|
||||
}
|
||||
|
||||
/** The only writer of policy-derived store state, so the bits cannot drift. */
|
||||
private applyPolicyToStore(store: SessionSearchStore): void {
|
||||
store.setHistoryDays(this.policy.historyDays)
|
||||
store.setAcceptingWrites(!this.policy.paused)
|
||||
store.indexing.setPaused(this.policy.paused === true)
|
||||
}
|
||||
|
||||
/** Waits for the aborted backfill so its last parse cannot write to a closed store. */
|
||||
@@ -247,9 +253,6 @@ export class SessionSearchService {
|
||||
}
|
||||
} finally {
|
||||
this.stopping = false
|
||||
if (options.keepStore) {
|
||||
this.store?.setAcceptingWrites(!this.policy.paused)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -260,7 +263,11 @@ export class SessionSearchService {
|
||||
if (!store) {
|
||||
return
|
||||
}
|
||||
registerSessionSearchIndexSink(null)
|
||||
// Why: shutdown is async, so a replacement service may already own the sink;
|
||||
// clearing it unconditionally would silently stop feeding the new index.
|
||||
if (getSessionSearchIndexSink() === store) {
|
||||
registerSessionSearchIndexSink(null)
|
||||
}
|
||||
store.close()
|
||||
}
|
||||
|
||||
@@ -271,7 +278,7 @@ export class SessionSearchService {
|
||||
return
|
||||
}
|
||||
try {
|
||||
await this.parseAll(this.withinHistory(stale), signal)
|
||||
await this.parseAll(this.withinHistory(stale), { signal })
|
||||
} catch (error) {
|
||||
// Why: a cancelled search (the renderer retires them per keystroke) must
|
||||
// not lose the queue; whatever did not get parsed goes back for next time.
|
||||
@@ -312,7 +319,7 @@ export class SessionSearchService {
|
||||
recordSearchDiscovered(store, discoveries, issues)
|
||||
const eligible = this.withinHistory(candidates)
|
||||
store.indexing.discovered(eligible.length, issues.length)
|
||||
await this.parseAll(eligible, signal, { yieldToSearches: signal })
|
||||
await this.parseAll(eligible, { signal, backfillSignal: signal })
|
||||
store.indexing.finish()
|
||||
store.setBackfillState('complete')
|
||||
} catch (error) {
|
||||
@@ -322,20 +329,26 @@ export class SessionSearchService {
|
||||
}
|
||||
}
|
||||
|
||||
/** `backfillSignal` marks the long tail: those files report progress and yield to searches. */
|
||||
private async parseAll(
|
||||
candidates: SessionFileCandidate[],
|
||||
signal?: AbortSignal,
|
||||
options: { yieldToSearches?: AbortSignal } = {}
|
||||
options: { signal?: AbortSignal; backfillSignal?: AbortSignal }
|
||||
): Promise<void> {
|
||||
if (this.store) {
|
||||
await parseSearchCandidates(
|
||||
this.store,
|
||||
candidates,
|
||||
signal,
|
||||
options.yieldToSearches
|
||||
? () => this.waitForIdleSearches(options.yieldToSearches!)
|
||||
: undefined
|
||||
)
|
||||
const store = this.store
|
||||
const backfillSignal = options.backfillSignal
|
||||
if (!store) {
|
||||
return
|
||||
}
|
||||
await parseSearchCandidates(store, candidates, {
|
||||
signal: options.signal,
|
||||
...(backfillSignal
|
||||
? {
|
||||
onFileProcessed: async (failed: boolean) => {
|
||||
store.indexing.processed(failed)
|
||||
await this.waitForIdleSearches(backfillSignal)
|
||||
}
|
||||
}
|
||||
: {})
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,33 +0,0 @@
|
||||
import type { AiVaultAgent } from '../../shared/ai-vault-types'
|
||||
import type { AiVaultSearchHit } from '../../shared/ai-vault-search-types'
|
||||
|
||||
export type SessionRow = {
|
||||
id: number
|
||||
agent: AiVaultAgent
|
||||
session_id: string
|
||||
file_path: string
|
||||
codex_home: string | null
|
||||
title: string
|
||||
cwd: string | null
|
||||
branch: string | null
|
||||
updated_at: string | null
|
||||
message_count: number
|
||||
resume_command: string
|
||||
content_hash: string | null
|
||||
content_hash_count: number
|
||||
}
|
||||
|
||||
export function sessionFields(session: SessionRow): Omit<AiVaultSearchHit, 'score' | 'evidence'> {
|
||||
return {
|
||||
agent: session.agent,
|
||||
sessionId: session.session_id,
|
||||
filePath: session.file_path,
|
||||
codexHome: session.codex_home,
|
||||
title: session.title,
|
||||
cwd: session.cwd,
|
||||
branch: session.branch,
|
||||
updatedAt: session.updated_at,
|
||||
messageCount: session.message_count,
|
||||
resumeCommand: session.resume_command
|
||||
}
|
||||
}
|
||||
@@ -58,7 +58,7 @@ it('checks individual OpenCode identities when several sessions share one databa
|
||||
}
|
||||
})
|
||||
|
||||
it('omits unreadable sources without treating them as confirmed deletion', async () => {
|
||||
it('keeps unreadable sources as hits and flags them instead of dropping them', async () => {
|
||||
const root = await mkdtemp(join(tmpdir(), 'orca-search-unreadable-'))
|
||||
try {
|
||||
const filePath = join(root, 'opencode.db')
|
||||
@@ -73,7 +73,8 @@ it('omits unreadable sources without treating them as confirmed deletion', async
|
||||
} as AiVaultSearchResult,
|
||||
invalidate
|
||||
)
|
||||
expect(result).toMatchObject({ hits: [], sourceUnavailableFiles: 1 })
|
||||
expect(result.hits.map((hit) => hit.sessionId)).toEqual(['fixture'])
|
||||
expect(result).toMatchObject({ sourceUnavailableFiles: 1 })
|
||||
expect(invalidate).not.toHaveBeenCalled()
|
||||
} finally {
|
||||
await rm(root, { recursive: true, force: true })
|
||||
|
||||
@@ -100,7 +100,7 @@ async function checkSearchSources(
|
||||
throwIfSignalAborted(signal)
|
||||
}
|
||||
|
||||
/** Refill limited results after filtering, without deleting unreadable sources or unbounded scans. */
|
||||
/** Refill limited results after dropping deleted sources, without unbounded scans. */
|
||||
export async function searchPresentSessionSources(
|
||||
args: AiVaultSearchArgs,
|
||||
search: (args: AiVaultSearchArgs) => AiVaultSearchResult,
|
||||
@@ -120,7 +120,10 @@ export async function searchPresentSessionSources(
|
||||
const result = search({ ...args, limit })
|
||||
durationMs += result.durationMs
|
||||
await checkSearchSources(result.hits, known, unavailable, invalidate, signal)
|
||||
const hits = result.hits.filter((hit) => known.get(candidatePath(hit)) === 'present')
|
||||
// Why: loss of contact is not evidence of absence (see
|
||||
// docs/reference/ssh-execution-boundary.md). Only a proven deletion drops a
|
||||
// hit; an unreadable source is still shown, counted in sourceUnavailableFiles.
|
||||
const hits = result.hits.filter((hit) => known.get(candidatePath(hit)) !== 'missing')
|
||||
const missing = result.hits.some((hit) => known.get(candidatePath(hit)) === 'missing')
|
||||
const exhausted = result.hits.length < limit && !missing
|
||||
const budgetExhausted =
|
||||
|
||||
@@ -4,7 +4,7 @@ import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import * as transcriptFs from '../native-chat/wsl-transcript-fs-access'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { stagedWriteUpdate } from './session-search-staged-write-fixtures'
|
||||
import { stagedWriteUpdate } from './session-search-staged-write-test-fixture'
|
||||
import { searchPresentSessionSources } from './session-search-source-presence'
|
||||
import type { AiVaultSearchArgs, AiVaultSearchResult } from '../../shared/ai-vault-search-types'
|
||||
|
||||
@@ -49,7 +49,7 @@ it('refills a deleted top hit in the first query with a real index', async () =>
|
||||
expect(result.omittedHits).toBeUndefined()
|
||||
})
|
||||
|
||||
it('refills past an unreadable top hit without deleting it or recounting it', async () => {
|
||||
it('keeps an unreadable top hit instead of refilling past it or deleting it', async () => {
|
||||
const { store, search, invalidate } = await fixture()
|
||||
const args = { query: 'refillneedle', limit: 1 }
|
||||
const top = store.search(args).hits[0]
|
||||
@@ -63,15 +63,15 @@ it('refills past an unreadable top hit without deleting it or recounting it', as
|
||||
return stat(path, priority, signal)
|
||||
})
|
||||
const result = await searchPresentSessionSources(args, search, invalidate)
|
||||
expect(result.hits).toHaveLength(1)
|
||||
expect(result.hits[0].filePath).not.toBe(top.filePath)
|
||||
// Loss of contact is not evidence of absence: the hit stays, flagged.
|
||||
expect(result.hits.map((hit) => hit.filePath)).toEqual([top.filePath])
|
||||
expect(result.sourceUnavailableFiles).toBe(1)
|
||||
expect(invalidate).not.toHaveBeenCalled()
|
||||
expect(store.search({ query: 'refillneedle' }).hits).toHaveLength(2)
|
||||
expect(probe.mock.calls.filter(([path]) => path === top.filePath)).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('bounds refill and reports omissions when unavailable hits consume the budget', async () => {
|
||||
it('bounds refill and reports omissions when deleted hits consume the budget', async () => {
|
||||
const { store } = await fixture()
|
||||
const base = store.search({ query: 'refillneedle' })
|
||||
const source = base.hits[0]
|
||||
@@ -80,7 +80,7 @@ it('bounds refill and reports omissions when unavailable hits consume the budget
|
||||
filePath: join(source.filePath, String(i))
|
||||
}))
|
||||
vi.spyOn(transcriptFs, 'wslGatedStat').mockRejectedValue(
|
||||
Object.assign(new Error('denied'), { code: 'EACCES' })
|
||||
Object.assign(new Error('gone'), { code: 'ENOENT' })
|
||||
)
|
||||
const search = vi.fn((args: AiVaultSearchArgs) => ({ ...base, hits: hits.slice(0, args.limit) }))
|
||||
const invalidate = vi.fn()
|
||||
@@ -91,8 +91,9 @@ it('bounds refill and reports omissions when unavailable hits consume the budget
|
||||
)
|
||||
expect(search).toHaveBeenCalledTimes(4)
|
||||
expect(search.mock.calls.map(([args]) => args.limit)).toEqual([1, 2, 4, 8])
|
||||
expect(result).toMatchObject({ hits: [], omittedHits: 8, sourceUnavailableFiles: 8 })
|
||||
expect(invalidate).not.toHaveBeenCalled()
|
||||
expect(result).toMatchObject({ hits: [], omittedHits: 8 })
|
||||
expect(result.sourceUnavailableFiles).toBeUndefined()
|
||||
expect(invalidate).toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('does not invalidate a WSL source when its share reports ENOENT', async () => {
|
||||
@@ -106,7 +107,7 @@ it('does not invalidate a WSL source when its share reports ENOENT', async () =>
|
||||
const invalidate = vi.fn()
|
||||
expect(
|
||||
await searchPresentSessionSources({ query: 'refillneedle' }, () => result, invalidate)
|
||||
).toMatchObject({ hits: [], sourceUnavailableFiles: 1 })
|
||||
).toMatchObject({ hits: [{ filePath }], sourceUnavailableFiles: 1 })
|
||||
expect(invalidate).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
import type { SessionSearchIndexUpdate } from '../ai-vault/session-search-capture'
|
||||
|
||||
export function stagedWriteUpdate(
|
||||
text: string,
|
||||
count: number,
|
||||
mode: 'append' | 'replace' = 'replace'
|
||||
): SessionSearchIndexUpdate {
|
||||
const at = new Date().toISOString()
|
||||
return {
|
||||
candidate: {
|
||||
agent: 'claude',
|
||||
codexHome: null,
|
||||
file: { path: 'synthetic-transcript', mtimeMs: Date.now(), modifiedAt: at, sizeBytes: 2 }
|
||||
},
|
||||
session: {
|
||||
id: 'fixture',
|
||||
executionHostId: 'local',
|
||||
agent: 'claude',
|
||||
sessionId: 'fixture',
|
||||
title: text,
|
||||
cwd: '/fixture',
|
||||
branch: null,
|
||||
model: null,
|
||||
filePath: 'synthetic-transcript',
|
||||
codexHome: null,
|
||||
createdAt: at,
|
||||
updatedAt: at,
|
||||
modifiedAt: at,
|
||||
messageCount: count,
|
||||
totalTokens: 0,
|
||||
previewMessages: [],
|
||||
queuedMessageCount: 0,
|
||||
subagentTranscriptCount: 0,
|
||||
resumeCommand: '',
|
||||
subagent: null
|
||||
},
|
||||
mode,
|
||||
messages: Array.from({ length: count }, () => ({ role: 'user', text, timestamp: null })),
|
||||
previousByteOffset: mode === 'append' ? 1 : 0,
|
||||
byteOffset: mode === 'append' ? 2 : 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
import { mkdtemp } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import type {
|
||||
SessionSearchIndexResult,
|
||||
SessionSearchIndexUpdate
|
||||
} from '../ai-vault/session-search-capture'
|
||||
import type { AiVaultSession } from '../../shared/ai-vault-types'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { openSessionSearchDatabase } from './session-search-schema'
|
||||
|
||||
/** Carries `session`/`byteOffset` alongside the write so assertions can read the expected result. */
|
||||
export function stagedWriteUpdate(
|
||||
text: string,
|
||||
count: number,
|
||||
mode: 'append' | 'replace' = 'replace'
|
||||
): SessionSearchIndexUpdate & { result: Promise<SessionSearchIndexResult> } {
|
||||
const at = new Date().toISOString()
|
||||
const session: AiVaultSession = {
|
||||
id: 'fixture',
|
||||
executionHostId: 'local',
|
||||
agent: 'claude',
|
||||
sessionId: 'fixture',
|
||||
title: text,
|
||||
cwd: '/fixture',
|
||||
branch: null,
|
||||
model: null,
|
||||
filePath: 'synthetic-transcript',
|
||||
codexHome: null,
|
||||
createdAt: at,
|
||||
updatedAt: at,
|
||||
modifiedAt: at,
|
||||
messageCount: count,
|
||||
totalTokens: 0,
|
||||
previewMessages: [],
|
||||
queuedMessageCount: 0,
|
||||
subagentTranscriptCount: 0,
|
||||
resumeCommand: '',
|
||||
subagent: null
|
||||
}
|
||||
const update: SessionSearchIndexUpdate & { result: Promise<SessionSearchIndexResult> } = {
|
||||
candidate: {
|
||||
agent: 'claude',
|
||||
codexHome: null,
|
||||
file: { path: 'synthetic-transcript', mtimeMs: Date.now(), modifiedAt: at, sizeBytes: 2 }
|
||||
},
|
||||
session,
|
||||
mode,
|
||||
messages: Array.from({ length: count }, () => ({ role: 'user', text, timestamp: null })),
|
||||
previousByteOffset: mode === 'append' ? 1 : 0,
|
||||
byteOffset: mode === 'append' ? 2 : 1,
|
||||
// Why: callers retarget `session`/`byteOffset` after construction, so the
|
||||
// promise the writer awaits must read them then, not at build time.
|
||||
result: Promise.resolve().then(() => ({
|
||||
session: update.session,
|
||||
byteOffset: update.byteOffset
|
||||
}))
|
||||
}
|
||||
return update
|
||||
}
|
||||
|
||||
export type SessionSearchIndexFile = {
|
||||
path: string
|
||||
/** The store keeps its own connection private, so row assertions need this one. */
|
||||
db: SyncDatabase
|
||||
close: () => Promise<void>
|
||||
}
|
||||
|
||||
/** An on-disk index: `:memory:` is per-connection, so a second reader needs a real file. */
|
||||
export async function openSessionSearchIndexFile(name: string): Promise<SessionSearchIndexFile> {
|
||||
const root = await mkdtemp(join(tmpdir(), `${name}-`))
|
||||
const path = join(root, 'index.sqlite')
|
||||
const db = openSessionSearchDatabase(path)
|
||||
let open = true
|
||||
return {
|
||||
path,
|
||||
db,
|
||||
close: async () => {
|
||||
if (open) {
|
||||
open = false
|
||||
db.close()
|
||||
}
|
||||
await removeTree(root)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,32 +1,38 @@
|
||||
import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import { mkdtemp } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { expect, it } from 'vitest'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { SessionSearchIndexWriter, SEARCH_WRITE_ROWS_PER_STEP } from './session-search-index-writer'
|
||||
import { stagedWriteUpdate as update } from './session-search-staged-write-fixtures'
|
||||
import {
|
||||
openSessionSearchIndexFile,
|
||||
stagedWriteUpdate as update
|
||||
} from './session-search-staged-write-test-fixture'
|
||||
|
||||
function stagedRows(db: { prepare: (sql: string) => { get: () => unknown } }): number {
|
||||
return (
|
||||
db
|
||||
.prepare(
|
||||
'SELECT count(*) AS n FROM messages WHERE batch_id IN (SELECT id FROM search_write_batches)'
|
||||
)
|
||||
.get() as { n: number }
|
||||
).n
|
||||
}
|
||||
|
||||
function messageCount(db: { prepare: (sql: string) => { get: () => unknown } }): number {
|
||||
return (db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n
|
||||
}
|
||||
|
||||
it.each(['replace', 'append'] as const)(
|
||||
'publishes a large %s atomically after bounded steps',
|
||||
async (mode) => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const index = await openSessionSearchIndexFile('ss-staged-publish')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
try {
|
||||
await writer.apply(update('oldneedle', 1))
|
||||
let steps = 0,
|
||||
previous = 0
|
||||
const applied = await writer.apply(
|
||||
update('newneedle', 1000, mode),
|
||||
() => true,
|
||||
async () => {
|
||||
const rows = (
|
||||
store.db
|
||||
.prepare(
|
||||
'SELECT count(*) AS n FROM messages WHERE batch_id IN (SELECT id FROM search_write_batches WHERE published=0)'
|
||||
)
|
||||
.get() as { n: number }
|
||||
).n
|
||||
const applied = await writer.apply(update('newneedle', 1000, mode), {
|
||||
yieldStep: async () => {
|
||||
const rows = stagedRows(index.db)
|
||||
expect(rows - previous).toBeLessThanOrEqual(SEARCH_WRITE_ROWS_PER_STEP)
|
||||
previous = rows
|
||||
steps++
|
||||
@@ -34,18 +40,17 @@ it.each(['replace', 'append'] as const)(
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1)
|
||||
expect(writer.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(1)
|
||||
}
|
||||
)
|
||||
})
|
||||
expect(applied).toBe(true)
|
||||
expect(steps).toBe(8)
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1)
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(mode === 'append' ? 1 : 0)
|
||||
expect(store.search({ query: 'newneedle' }).hits[0].title).toBe('newneedle')
|
||||
await store.purgeOlderThan(null)
|
||||
expect(
|
||||
(store.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n
|
||||
).toBe(mode === 'append' ? 1001 : 1000)
|
||||
expect(messageCount(index.db)).toBe(mode === 'append' ? 1001 : 1000)
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
}
|
||||
)
|
||||
@@ -53,93 +58,120 @@ it.each(['replace', 'append'] as const)(
|
||||
it.each(['replace', 'append'] as const)(
|
||||
'recovers an interrupted %s without publishing rows or advancing its cursor',
|
||||
async (mode) => {
|
||||
const root = await mkdtemp(join(tmpdir(), 'ss-staged-'))
|
||||
const path = join(root, 'index.sqlite')
|
||||
let store = new SessionSearchStore(path),
|
||||
open = true
|
||||
const index = await openSessionSearchIndexFile('ss-staged-recover')
|
||||
let store = new SessionSearchStore(index.path)
|
||||
let open = true
|
||||
try {
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
await writer.apply(update('oldneedle', 1))
|
||||
expect(
|
||||
await writer.apply(
|
||||
update('newneedle', 1000, mode),
|
||||
() => open,
|
||||
async () => {
|
||||
await writer.apply(update('newneedle', 1000, mode), {
|
||||
active: () => open,
|
||||
yieldStep: async () => {
|
||||
store.close()
|
||||
open = false
|
||||
}
|
||||
)
|
||||
})
|
||||
).toBe(false)
|
||||
store = new SessionSearchStore(path)
|
||||
store = new SessionSearchStore(index.path)
|
||||
open = true
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0)
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1)
|
||||
expect(store.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(1)
|
||||
await store.purgeOlderThan(null)
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1 })
|
||||
expect(messageCount(index.db)).toBe(1)
|
||||
await store.apply(update('newneedle', 1000, mode))
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1)
|
||||
} finally {
|
||||
if (open) {
|
||||
store.close()
|
||||
}
|
||||
await removeTree(root)
|
||||
await index.close()
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
it('does not resurrect a newly invalidated file or retain a cancelled append', async () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const index = await openSessionSearchIndexFile('ss-staged-invalidated')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
try {
|
||||
expect(
|
||||
await writer.apply(
|
||||
update('newneedle', 1000),
|
||||
() => true,
|
||||
async () => {
|
||||
await writer.apply(update('newneedle', 1000), {
|
||||
yieldStep: async () => {
|
||||
writer.removeFile('synthetic-transcript')
|
||||
}
|
||||
)
|
||||
})
|
||||
).toBe(false)
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0)
|
||||
await store.purgeOlderThan(null)
|
||||
await writer.apply(update('oldneedle', 1))
|
||||
let accepted = true
|
||||
expect(
|
||||
await writer.apply(
|
||||
update('newneedle', 1000, 'append'),
|
||||
() => accepted,
|
||||
async () => {
|
||||
await writer.apply(update('newneedle', 1000, 'append'), {
|
||||
active: () => accepted,
|
||||
yieldStep: async () => {
|
||||
accepted = false
|
||||
},
|
||||
() => true
|
||||
)
|
||||
available: () => true
|
||||
})
|
||||
).toBe(false)
|
||||
await store.purgeOlderThan(null)
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1 })
|
||||
expect(messageCount(index.db)).toBe(1)
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1)
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('does not suggest unpublished vocabulary while a large update is being written', async () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const index = await openSessionSearchIndexFile('ss-staged-vocab')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
try {
|
||||
await writer.apply(update('oldneedle', 1))
|
||||
await writer.apply(
|
||||
update('coalesces', 1000, 'append'),
|
||||
() => true,
|
||||
async () => {
|
||||
await writer.apply(update('coalesces', 1000, 'append'), {
|
||||
yieldStep: async () => {
|
||||
const result = store.search({ query: 'coalescs' })
|
||||
expect(result.hits).toHaveLength(0)
|
||||
expect(result.repairedTerms).toBeUndefined()
|
||||
}
|
||||
)
|
||||
})
|
||||
expect(store.search({ query: 'coalescs' }).repairedTerms).toEqual(['coalesces'])
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('keeps published rows visible when a later batch reuses the freed id', async () => {
|
||||
const index = await openSessionSearchIndexFile('ss-staged-batch-reuse')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
const inFlightBatchId = (): number =>
|
||||
(index.db.prepare('SELECT id FROM search_write_batches').get() as { id: number }).id
|
||||
try {
|
||||
let published: number | null = null
|
||||
await writer.apply(update('oldneedle', 1000), {
|
||||
yieldStep: async () => {
|
||||
published ??= inFlightBatchId()
|
||||
}
|
||||
})
|
||||
let reused: number | null = null
|
||||
await writer.apply(update('newneedle', 1000, 'append'), {
|
||||
yieldStep: async () => {
|
||||
reused ??= inFlightBatchId()
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1)
|
||||
}
|
||||
})
|
||||
// Without this the test proves nothing: SQLite hands the freed rowid straight back.
|
||||
expect(reused).toBe(published)
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1)
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1)
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -153,7 +185,7 @@ it('keeps published data and queues recovery when an append cursor is stale', as
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1)
|
||||
expect(store.search({ query: 'staleneedle' }).hits).toHaveLength(0)
|
||||
expect(store.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(2)
|
||||
expect(store.staleCount).toBe(1)
|
||||
expect(store.coverage().filesPending).toBe(1)
|
||||
} finally {
|
||||
store.close()
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import {
|
||||
registerSessionSearchIndexSink,
|
||||
withSessionSearchIndexRequired
|
||||
} from '../ai-vault/session-search-capture'
|
||||
import SyncDatabase from '../sqlite/sync-database'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import {
|
||||
assistantRecord,
|
||||
@@ -16,11 +17,13 @@ import {
|
||||
} from './session-search-transcript-fixtures'
|
||||
let tempRoots: string[] = []
|
||||
let store: SessionSearchStore
|
||||
let databasePath: string
|
||||
|
||||
beforeEach(async () => {
|
||||
resetSessionParseCacheForTests()
|
||||
const root = await makeTempDir()
|
||||
store = new SessionSearchStore(join(root, 'index.sqlite'), (error) => {
|
||||
databasePath = join(root, 'index.sqlite')
|
||||
store = new SessionSearchStore(databasePath, (error) => {
|
||||
throw error
|
||||
})
|
||||
registerSessionSearchIndexSink(store)
|
||||
@@ -279,14 +282,19 @@ describe('SessionSearchStore', () => {
|
||||
await writeFile(path, `${userRecord(0, `padding ${filler}`, id)}\n`)
|
||||
await parse(path)
|
||||
}
|
||||
const pageCount = (): number => Number(store.db.pragma('page_count', { simple: true }))
|
||||
const before = pageCount()
|
||||
const reader = new SyncDatabase(databasePath, { readonly: true })
|
||||
try {
|
||||
const pageCount = (): number => Number(reader.pragma('page_count', { simple: true }))
|
||||
const before = pageCount()
|
||||
|
||||
await store.purgeOlderThan(Date.now() + 60_000)
|
||||
await store.purgeOlderThan(Date.now() + 60_000)
|
||||
|
||||
expect(store.coverage().sessionsIndexed).toBe(0)
|
||||
expect(Number(store.db.pragma('freelist_count', { simple: true }))).toBe(0)
|
||||
expect(pageCount()).toBeLessThan(before)
|
||||
expect(store.coverage().sessionsIndexed).toBe(0)
|
||||
expect(Number(reader.pragma('freelist_count', { simple: true }))).toBe(0)
|
||||
expect(pageCount()).toBeLessThan(before)
|
||||
} finally {
|
||||
reader.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('warms once and survives a close mid-way', async () => {
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
import { SessionSearchIndexingProgress } from './session-search-indexing-progress'
|
||||
import { SessionSearchMaintenance } from './session-search-maintenance'
|
||||
import { recoverSearchWrites } from './session-search-write-recovery'
|
||||
import { compactSessionSearchIndex } from './session-search-index-compaction'
|
||||
import { logSessionSearchQuery } from './session-search-query-log'
|
||||
import { warmSessionSearchPages } from './session-search-page-warmup'
|
||||
import { recoverSearchWrites } from './session-search-pending-deletes'
|
||||
import { deleteExpiredSearchFiles } from './session-search-retention-delete'
|
||||
import type { AiVaultAgent } from '../../shared/ai-vault-types'
|
||||
import { aiVaultSearchHistoryCutoffMs } from '../../shared/ai-vault-search-settings'
|
||||
@@ -20,19 +22,26 @@ import type {
|
||||
import type { SessionFileCandidate } from '../ai-vault/session-scanner-types'
|
||||
import { SessionSearchIndexWriter, type SessionSearchMetadata } from './session-search-index-writer'
|
||||
import { SessionSearchQuery } from './session-search-query'
|
||||
import { openSessionSearchDatabase } from './session-search-schema'
|
||||
import {
|
||||
openSessionSearchDatabase,
|
||||
VISIBLE_MESSAGES,
|
||||
VISIBLE_SESSIONS
|
||||
} from './session-search-schema'
|
||||
|
||||
export type SessionSearchBackfillState = 'idle' | 'running' | 'complete'
|
||||
|
||||
type ProviderDiscovery = { files: number; parseFailures: number; scanIssues: number }
|
||||
|
||||
export type SessionSearchStoreOptions = {
|
||||
/** The WAL backlog a staging write refuses to grow past. Only tests narrow it. */
|
||||
walBudgetBytes?: number
|
||||
}
|
||||
|
||||
/** Owns the index database: the scanner writes through it, search reads from it. */
|
||||
export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
readonly streamingCapture = true
|
||||
readonly indexing = new SessionSearchIndexingProgress()
|
||||
private writeEpoch = 0
|
||||
/** Exposed for tests that assert on file-level state (page counts). */
|
||||
readonly db: SyncDatabase
|
||||
private readonly db: SyncDatabase
|
||||
private readonly writer: SessionSearchIndexWriter
|
||||
private readonly query: SessionSearchQuery
|
||||
private backfill: SessionSearchBackfillState = 'idle'
|
||||
@@ -43,7 +52,7 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
null
|
||||
private cleanupRequested = false
|
||||
private cleanup: Promise<void> | null = null
|
||||
private readonly maintenance: SessionSearchMaintenance
|
||||
private warmed: Promise<void> | null = null
|
||||
private lastIndexedAt: string | null = null
|
||||
private applyFailures = 0
|
||||
private readonly stale = new Map<string, SessionFileCandidate>()
|
||||
@@ -55,13 +64,13 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
console.warn(
|
||||
'[ai-vault-search] index write failed:',
|
||||
error instanceof Error ? error.name : 'IndexError'
|
||||
)
|
||||
),
|
||||
options: SessionSearchStoreOptions = {}
|
||||
) {
|
||||
this.db = openSessionSearchDatabase(path)
|
||||
recoverSearchWrites(this.db)
|
||||
this.writer = new SessionSearchIndexWriter(this.db)
|
||||
this.writer = new SessionSearchIndexWriter(this.db, options.walBudgetBytes)
|
||||
this.query = new SessionSearchQuery(this.db)
|
||||
this.maintenance = new SessionSearchMaintenance(this.db, () => this.closed, this.onError)
|
||||
}
|
||||
|
||||
indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null {
|
||||
@@ -119,15 +128,13 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
const epoch = this.writeEpoch
|
||||
const finish = this.indexing.beginWrite()
|
||||
try {
|
||||
const applied = await this.writer.apply(
|
||||
update,
|
||||
() =>
|
||||
const applied = await this.writer.apply(update, {
|
||||
active: () =>
|
||||
!update.signal?.aborted &&
|
||||
epoch === this.writeEpoch &&
|
||||
this.acceptsCandidate(update.candidate),
|
||||
undefined,
|
||||
() => !this.closed
|
||||
)
|
||||
available: () => !this.closed
|
||||
})
|
||||
if (!applied) {
|
||||
this.markStale(update.candidate)
|
||||
return
|
||||
@@ -164,10 +171,6 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
return candidates
|
||||
}
|
||||
|
||||
get staleCount(): number {
|
||||
return this.stale.size
|
||||
}
|
||||
|
||||
removeFile(path: string): void {
|
||||
this.providerCounts = null
|
||||
this.stale.delete(path)
|
||||
@@ -221,7 +224,7 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
}
|
||||
)
|
||||
if (!this.closed && !signal?.aborted) {
|
||||
await this.maintenance.compact(signal)
|
||||
await compactSessionSearchIndex(this.db, () => this.closed || signal?.aborted === true)
|
||||
}
|
||||
} catch (error) {
|
||||
if (!this.closed) {
|
||||
@@ -231,7 +234,10 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
}
|
||||
|
||||
warm(): Promise<void> {
|
||||
return this.maintenance.warm()
|
||||
this.warmed ??= warmSessionSearchPages(this.db, () => this.closed).catch((error) =>
|
||||
this.onError(error)
|
||||
)
|
||||
return this.warmed
|
||||
}
|
||||
|
||||
setBackfillState(state: SessionSearchBackfillState): void {
|
||||
@@ -253,7 +259,16 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
const startedAt = performance.now()
|
||||
const execution = this.query.execute(args, aiVaultSearchHistoryCutoffMs(this.historyDays))
|
||||
const durationMs = performance.now() - startedAt
|
||||
this.maintenance.logQuery(args.query, execution.route, execution.hits.length, durationMs)
|
||||
try {
|
||||
logSessionSearchQuery(this.db, {
|
||||
query: args.query,
|
||||
route: execution.route,
|
||||
hits: execution.hits.length,
|
||||
durationMs
|
||||
})
|
||||
} catch (error) {
|
||||
this.onError(error)
|
||||
}
|
||||
return {
|
||||
hits: execution.hits,
|
||||
route: execution.route,
|
||||
@@ -267,9 +282,7 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
const providers = (this.providerCounts ??= this.db
|
||||
.prepare(
|
||||
`SELECT s.agent AS agent, COUNT(DISTINCT s.id) AS sessions, COUNT(m.id) AS messages
|
||||
FROM sessions s LEFT JOIN messages m ON m.session_row_id = s.id
|
||||
AND (m.batch_id IS NULL OR m.batch_id NOT IN (SELECT id FROM search_write_batches WHERE published=0))
|
||||
WHERE s.index_ready=1 AND s.id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL)
|
||||
FROM ${VISIBLE_SESSIONS} s LEFT JOIN ${VISIBLE_MESSAGES} m ON m.session_row_id = s.id
|
||||
GROUP BY s.agent ORDER BY s.agent`
|
||||
)
|
||||
.all() as { agent: AiVaultAgent; sessions: number; messages: number }[])
|
||||
|
||||
@@ -6,13 +6,17 @@ import {
|
||||
import { captureIndexedSessionParse } from '../ai-vault/session-search-indexed-parse'
|
||||
import { SessionSearchIndexWriter } from './session-search-index-writer'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { stagedWriteUpdate } from './session-search-staged-write-fixtures'
|
||||
import {
|
||||
openSessionSearchIndexFile,
|
||||
stagedWriteUpdate
|
||||
} from './session-search-staged-write-test-fixture'
|
||||
|
||||
it.each(['publish', 'reject', 'throw', 'cancel'] as const)(
|
||||
'streams capture with atomic %s',
|
||||
async (outcome) => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const index = await openSessionSearchIndexFile('ss-streaming')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
const original = stagedWriteUpdate('oldneedle', 1)
|
||||
const replacement = stagedWriteUpdate('newneedle', 600)
|
||||
let active = true
|
||||
@@ -20,16 +24,10 @@ it.each(['publish', 'reject', 'throw', 'cancel'] as const)(
|
||||
await writer.apply(original)
|
||||
const parse = captureIndexedSessionParse(
|
||||
{
|
||||
streamingCapture: true,
|
||||
indexedFile: () => null,
|
||||
markStale: () => {},
|
||||
apply: async (update) => {
|
||||
await writer.apply(
|
||||
update,
|
||||
() => active,
|
||||
undefined,
|
||||
() => true
|
||||
)
|
||||
await writer.apply(update, { active: () => active, available: () => true })
|
||||
}
|
||||
},
|
||||
replacement,
|
||||
@@ -40,7 +38,7 @@ it.each(['publish', 'reject', 'throw', 'cancel'] as const)(
|
||||
if (i === 400) {
|
||||
expect(
|
||||
Number(
|
||||
(store.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n
|
||||
(index.db.prepare('SELECT count(*) AS n FROM messages').get() as { n: number }).n
|
||||
)
|
||||
).toBeGreaterThan(128)
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0)
|
||||
@@ -76,18 +74,18 @@ it.each(['publish', 'reject', 'throw', 'cancel'] as const)(
|
||||
expect(store.search({ query: 'newneedle' }).hits[0].title).toBe('newneedle')
|
||||
}
|
||||
await store.purgeOlderThan(null)
|
||||
expect(
|
||||
store.db.prepare('SELECT id FROM search_write_batches WHERE published=0').all()
|
||||
).toHaveLength(0)
|
||||
expect(index.db.prepare('SELECT id FROM search_write_batches').all()).toHaveLength(0)
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
it('waits for final metadata after the last message without mutating the write', async () => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
const writer = new SessionSearchIndexWriter(store.db)
|
||||
const index = await openSessionSearchIndexFile('ss-streaming-final')
|
||||
const store = new SessionSearchStore(index.path)
|
||||
const writer = new SessionSearchIndexWriter(index.db)
|
||||
const replacement = stagedWriteUpdate('finaltitle', 1)
|
||||
const { promise: result, resolve } = Promise.withResolvers<{
|
||||
session: typeof replacement.session
|
||||
@@ -117,5 +115,6 @@ it('waits for final metadata after the last message without mutating the write',
|
||||
} finally {
|
||||
resolve({ session: null, byteOffset: 0 })
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { openSessionSearchIndexFile } from './session-search-staged-write-test-fixture'
|
||||
import { SessionSearchTypoRepair } from './session-search-typo-repair'
|
||||
|
||||
describe('typo repair policy', () => {
|
||||
@@ -12,19 +12,19 @@ describe('typo repair policy', () => {
|
||||
{ input: 'calm', candidate: 'clam', copies: 2, exact: false, expected: null }
|
||||
])(
|
||||
'repairs $input to $expected with $copies postings (exact=$exact)',
|
||||
({ input, candidate, copies, exact, expected }) => {
|
||||
const store = new SessionSearchStore(':memory:')
|
||||
async ({ input, candidate, copies, exact, expected }) => {
|
||||
const index = await openSessionSearchIndexFile('ss-typo-policy')
|
||||
try {
|
||||
const insert = store.db.prepare('INSERT INTO messages_fts(user_text) VALUES (?)')
|
||||
const insert = index.db.prepare('INSERT INTO messages_fts(user_text) VALUES (?)')
|
||||
for (let i = 0; i < copies; i++) {
|
||||
insert.run(candidate)
|
||||
}
|
||||
if (exact) {
|
||||
insert.run(input)
|
||||
}
|
||||
expect(new SessionSearchTypoRepair(store.db).correct(input)).toBe(expected)
|
||||
expect(new SessionSearchTypoRepair(index.db).correct(input)).toBe(expected)
|
||||
} finally {
|
||||
store.close()
|
||||
await index.close()
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { quoteFtsTerm } from './session-search-query-planner'
|
||||
import { VISIBLE_MESSAGES, VISIBLE_SESSIONS } from './session-search-schema'
|
||||
|
||||
// Why: a query term with zero postings is usually a typo. The index's own
|
||||
// vocabulary (fts5vocab) is the dictionary, so repair needs no model and can
|
||||
@@ -43,13 +45,12 @@ export class SessionSearchTypoRepair {
|
||||
|
||||
constructor(db: SyncDatabase) {
|
||||
this.unpublished = db.prepare(
|
||||
'SELECT 1 FROM search_write_batches WHERE published=0 UNION ALL SELECT 1 FROM search_pending_deletes LIMIT 1'
|
||||
'SELECT 1 FROM search_write_batches UNION ALL SELECT 1 FROM search_pending_deletes LIMIT 1'
|
||||
)
|
||||
this.visiblePostings =
|
||||
db.prepare(`SELECT m.id FROM messages_fts JOIN messages m ON m.id=messages_fts.rowid
|
||||
JOIN sessions s ON s.id=m.session_row_id WHERE messages_fts MATCH ? AND s.index_ready=1
|
||||
AND s.id NOT IN (SELECT session_row_id FROM search_pending_deletes WHERE batch_id IS NULL)
|
||||
AND (m.batch_id IS NULL OR m.batch_id NOT IN (SELECT id FROM search_write_batches WHERE published=0)) LIMIT 2`)
|
||||
db.prepare(`SELECT m.id FROM messages_fts JOIN ${VISIBLE_MESSAGES} m ON m.id=messages_fts.rowid
|
||||
JOIN ${VISIBLE_SESSIONS} s ON s.id=m.session_row_id WHERE messages_fts MATCH ?
|
||||
LIMIT ${MIN_DOC_FREQUENCY}`)
|
||||
|
||||
this.exactMatch = db.prepare(
|
||||
'SELECT rowid FROM messages_fts WHERE messages_fts MATCH ? LIMIT 1'
|
||||
@@ -65,17 +66,21 @@ export class SessionSearchTypoRepair {
|
||||
}
|
||||
|
||||
hasPostings(term: string): boolean {
|
||||
if (this.unpublished.get()) {
|
||||
return this.visiblePostings.all(`"${term.replaceAll('"', '""')}"`).length > 0
|
||||
if (this.hasUnpublishedWrites()) {
|
||||
return this.visiblePostings.all(quoteFtsTerm(term)).length > 0
|
||||
}
|
||||
const row = this.documentFrequency.get(term.toLowerCase()) as VocabRow | undefined
|
||||
// unicode61 also folds Latin diacritics; raw vocabulary spelling alone can miss an exact hit.
|
||||
return (
|
||||
(row !== undefined && row.doc > 0) ||
|
||||
this.exactMatch.get(`"${term.replaceAll('"', '""')}"`) !== undefined
|
||||
(row !== undefined && row.doc > 0) || this.exactMatch.get(quoteFtsTerm(term)) !== undefined
|
||||
)
|
||||
}
|
||||
|
||||
/** A staged write is uncommitted, so `messages_vocab` can list a term no visible row has yet. */
|
||||
private hasUnpublishedWrites(): boolean {
|
||||
return this.unpublished.get() !== undefined
|
||||
}
|
||||
|
||||
/** Returns the closest indexed term, or null when `term` exists or nothing is close enough. */
|
||||
correct(term: string): string | null {
|
||||
const lowered = term.toLowerCase()
|
||||
@@ -87,32 +92,40 @@ export class SessionSearchTypoRepair {
|
||||
}
|
||||
// Two-letter prefix first (a typo rarely hits both), then the transposed
|
||||
// pair, then the bare first letter as the wide fallback.
|
||||
const unpublished = this.unpublished.get() !== undefined
|
||||
const prefixes = [lowered.slice(0, 2), lowered[1] + lowered[0], lowered[0]]
|
||||
let best: { term: string; score: number; doc: number } | null = null
|
||||
for (const prefix of prefixes) {
|
||||
for (const row of this.candidates(prefix, lowered.length)) {
|
||||
const score = similarity(lowered, row.term)
|
||||
if (score < MIN_SIMILARITY) {
|
||||
continue
|
||||
}
|
||||
if (
|
||||
unpublished &&
|
||||
this.visiblePostings.all(`"${row.term.replaceAll('"', '""')}"`).length < MIN_DOC_FREQUENCY
|
||||
) {
|
||||
continue
|
||||
}
|
||||
if (!best || score > best.score || (score === best.score && row.doc > best.doc)) {
|
||||
best = { term: row.term, score, doc: row.doc }
|
||||
}
|
||||
}
|
||||
if (best) {
|
||||
const best = this.closest(lowered, prefix)
|
||||
// Why the visibility probe is on the winner alone: ranking is pure CPU,
|
||||
// but each probe is an FTS MATCH, and during a backfill — exactly when
|
||||
// people search — one per candidate is thousands of queries per prefix.
|
||||
if (best && this.isVisible(best.term)) {
|
||||
return best.term
|
||||
}
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
private closest(lowered: string, prefix: string): { term: string; score: number } | null {
|
||||
let best: { term: string; score: number; doc: number } | null = null
|
||||
for (const row of this.candidates(prefix, lowered.length)) {
|
||||
const score = similarity(lowered, row.term)
|
||||
if (score < MIN_SIMILARITY) {
|
||||
continue
|
||||
}
|
||||
if (!best || score > best.score || (score === best.score && row.doc > best.doc)) {
|
||||
best = { term: row.term, score, doc: row.doc }
|
||||
}
|
||||
}
|
||||
return best
|
||||
}
|
||||
|
||||
private isVisible(term: string): boolean {
|
||||
return (
|
||||
!this.hasUnpublishedWrites() ||
|
||||
this.visiblePostings.all(quoteFtsTerm(term)).length >= MIN_DOC_FREQUENCY
|
||||
)
|
||||
}
|
||||
|
||||
private candidates(prefix: string, length: number): VocabRow[] {
|
||||
const last = prefix.charCodeAt(prefix.length - 1)
|
||||
const upper = prefix.slice(0, -1) + String.fromCharCode(last + 1)
|
||||
|
||||
@@ -1,65 +1,57 @@
|
||||
import { removeTree } from '../../shared/windows-transient-lock-removal'
|
||||
import { mkdtemp } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { expect, it, vi } from 'vitest'
|
||||
import * as Wal from './session-search-wal-budget'
|
||||
import { stagedWriteUpdate } from './session-search-staged-write-fixtures'
|
||||
import { expect, it } from 'vitest'
|
||||
import SyncDatabase from '../sqlite/sync-database'
|
||||
import {
|
||||
openSessionSearchIndexFile,
|
||||
stagedWriteUpdate
|
||||
} from './session-search-staged-write-test-fixture'
|
||||
import { SessionSearchStore } from './session-search-store'
|
||||
import { assertSearchWalBudget, SearchWalBackpressureError } from './session-search-wal-budget'
|
||||
|
||||
it('backpressures a pinned snapshot and resumes checkpoints after that reader releases', async () => {
|
||||
const root = await mkdtemp(join(tmpdir(), 'ss-wal-budget-'))
|
||||
const path = join(root, 'index.sqlite')
|
||||
const store = new SessionSearchStore(path)
|
||||
const reader = new SyncDatabase(path, { readonly: true })
|
||||
const index = await openSessionSearchIndexFile('ss-wal-budget')
|
||||
const reader = new SyncDatabase(index.path, { readonly: true })
|
||||
try {
|
||||
assertSearchWalBudget(store.db)
|
||||
assertSearchWalBudget(index.db)
|
||||
reader.exec('BEGIN')
|
||||
reader.prepare('SELECT count(*) FROM sessions').get()
|
||||
store.db
|
||||
index.db
|
||||
.prepare("INSERT INTO search_log(ts,query,route,hits,duration_ms) VALUES ('t',?,'or',0,0)")
|
||||
.run('synthetic'.repeat(10000))
|
||||
expect(() => assertSearchWalBudget(store.db, 4096)).toThrow(SearchWalBackpressureError)
|
||||
expect(() => assertSearchWalBudget(index.db, 4096)).toThrow(SearchWalBackpressureError)
|
||||
reader.exec('COMMIT')
|
||||
expect(() => assertSearchWalBudget(store.db, 4096)).not.toThrow()
|
||||
expect(() => assertSearchWalBudget(index.db, 4096)).not.toThrow()
|
||||
} finally {
|
||||
reader.close()
|
||||
store.close()
|
||||
await removeTree(root)
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('retains the old searchable generation and retries a backpressured write after reader release', async () => {
|
||||
const root = await mkdtemp(join(tmpdir(), 'ss-wal-write-'))
|
||||
const path = join(root, 'index.sqlite')
|
||||
const index = await openSessionSearchIndexFile('ss-wal-write')
|
||||
const errors: unknown[] = []
|
||||
const store = new SessionSearchStore(path, (error) => errors.push(error))
|
||||
const reader = new SyncDatabase(path, { readonly: true })
|
||||
const store = new SessionSearchStore(index.path, (error) => errors.push(error), {
|
||||
walBudgetBytes: 4096
|
||||
})
|
||||
const reader = new SyncDatabase(index.path, { readonly: true })
|
||||
try {
|
||||
await store.apply(stagedWriteUpdate('oldneedle', 1))
|
||||
reader.exec('BEGIN')
|
||||
reader.prepare('SELECT count(*) FROM messages').get()
|
||||
const actual = Wal.assertSearchWalBudget
|
||||
const spy = vi.spyOn(Wal, 'assertSearchWalBudget').mockImplementation((db) => actual(db, 4096))
|
||||
await store.apply(stagedWriteUpdate('newneedle', 1000, 'append'))
|
||||
expect(errors.some((error) => error instanceof SearchWalBackpressureError)).toBe(true)
|
||||
expect(store.staleCount).toBe(1)
|
||||
expect(store.coverage().filesPending).toBe(1)
|
||||
expect(store.search({ query: 'oldneedle' }).hits).toHaveLength(1)
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(0)
|
||||
expect(store.indexedFile('synthetic-transcript', null)?.byteOffset).toBe(1)
|
||||
reader.exec('COMMIT')
|
||||
spy.mockRestore()
|
||||
await store.apply(stagedWriteUpdate('newneedle', 1000, 'append'))
|
||||
await store.purgeOlderThan(null)
|
||||
expect(store.search({ query: 'newneedle' }).hits).toHaveLength(1)
|
||||
expect(store.staleCount).toBe(0)
|
||||
expect(store.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1001 })
|
||||
expect(store.coverage().filesPending).toBe(0)
|
||||
expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 1001 })
|
||||
} finally {
|
||||
vi.restoreAllMocks()
|
||||
reader.close()
|
||||
store.close()
|
||||
await removeTree(root)
|
||||
await index.close()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -7,10 +7,11 @@ export class SessionNewestFiles {
|
||||
private readonly limit: number
|
||||
|
||||
constructor(limit: number) {
|
||||
this.limit = limit === Infinity ? limit : Math.max(0, Math.trunc(limit) || 0)
|
||||
this.limit = Math.max(0, Math.trunc(limit) || 0)
|
||||
}
|
||||
|
||||
add(file: FileWithMtime): void {
|
||||
// The backfill enumerates with no limit; skip the insert search entirely.
|
||||
if (!Number.isFinite(this.limit)) {
|
||||
this.files.push(file)
|
||||
return
|
||||
@@ -42,7 +43,8 @@ export class SessionNewestFiles {
|
||||
return this.files.length
|
||||
}
|
||||
|
||||
/** The unbounded path appends in traversal order, so the sort is not redundant. */
|
||||
newest(): FileWithMtime[] {
|
||||
return this.files.sort((a, b) => b.mtimeMs - a.mtimeMs)
|
||||
return [...this.files].sort((a, b) => b.mtimeMs - a.mtimeMs)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,7 +2,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { mkdir, mkdtemp, readdir, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { walkSessionFiles } from './session-scanner-discovery'
|
||||
import { forEachSessionFile, walkSessionFiles } from './session-scanner-discovery'
|
||||
|
||||
let tempRoot: string | null = null
|
||||
|
||||
@@ -76,13 +76,14 @@ it('visits file contents before descending further without retaining paths', asy
|
||||
a.name.localeCompare(b.name)
|
||||
)
|
||||
})
|
||||
const retained = await walkSessionFiles(tempRoot, 'claude', [], {
|
||||
extensions: new Set(['.jsonl']),
|
||||
readDirectory,
|
||||
onFile: async (path) => {
|
||||
await forEachSessionFile(
|
||||
tempRoot,
|
||||
'claude',
|
||||
[],
|
||||
{ extensions: new Set(['.jsonl']), readDirectory },
|
||||
async (path) => {
|
||||
visited.push(path)
|
||||
}
|
||||
})
|
||||
expect(retained).toEqual([])
|
||||
)
|
||||
expect(visited).toHaveLength(2)
|
||||
})
|
||||
|
||||
@@ -21,12 +21,17 @@ export async function discoverFiles(args: {
|
||||
}): Promise<SessionFileDiscovery> {
|
||||
const files = new SessionNewestFiles(args.limit)
|
||||
try {
|
||||
await walkSessionFiles(args.rootDir, args.agent, args.issues, {
|
||||
extensions: new Set(args.extensions),
|
||||
signal: args.signal,
|
||||
filePredicate: args.filePredicate,
|
||||
directoryPredicate: args.directoryPredicate,
|
||||
onFile: async (path) => {
|
||||
await forEachSessionFile(
|
||||
args.rootDir,
|
||||
args.agent,
|
||||
args.issues,
|
||||
{
|
||||
extensions: new Set(args.extensions),
|
||||
signal: args.signal,
|
||||
filePredicate: args.filePredicate,
|
||||
directoryPredicate: args.directoryPredicate
|
||||
},
|
||||
async (path) => {
|
||||
args.signal?.throwIfAborted()
|
||||
try {
|
||||
const fileStat = await wslGatedStat(path, 'scan')
|
||||
@@ -53,8 +58,11 @@ export async function discoverFiles(args: {
|
||||
})
|
||||
}
|
||||
}
|
||||
})
|
||||
)
|
||||
} catch (err) {
|
||||
// Why: discoverAiVaultSessionSources fans out with Promise.all, so one
|
||||
// stalled distro would otherwise reject the whole vault scan — including
|
||||
// every healthy local agent. Contain it to this root.
|
||||
if (!(err instanceof WslTranscriptFsError)) {
|
||||
throw err
|
||||
}
|
||||
@@ -85,22 +93,39 @@ async function optionalContentDependencyStat(
|
||||
}
|
||||
}
|
||||
|
||||
export type SessionFileWalkOptions = {
|
||||
extensions: Set<string>
|
||||
filePredicate?: (path: string) => boolean
|
||||
// Return false to skip descending into a directory; depth 0 is a child of
|
||||
// rootDir, so pruned subtrees are never stat'd or parsed.
|
||||
directoryPredicate?: (name: string, depth: number) => boolean
|
||||
readDirectory?: (dirPath: string) => Promise<Dirent[]>
|
||||
signal?: AbortSignal
|
||||
}
|
||||
|
||||
/** Collecting form for callers that want every match; large scans use `forEachSessionFile`. */
|
||||
export async function walkSessionFiles(
|
||||
dirPath: string,
|
||||
agent: AiVaultAgent,
|
||||
issues: AiVaultScanIssue[],
|
||||
options: {
|
||||
extensions: Set<string>
|
||||
filePredicate?: (path: string) => boolean
|
||||
// Return false to skip descending into a directory; depth 0 is a child of
|
||||
// rootDir, so pruned subtrees are never stat'd or parsed.
|
||||
directoryPredicate?: (name: string, depth: number) => boolean
|
||||
readDirectory?: (dirPath: string) => Promise<Dirent[]>
|
||||
signal?: AbortSignal
|
||||
onFile?: (path: string) => Promise<void>
|
||||
},
|
||||
depth = 0
|
||||
options: SessionFileWalkOptions
|
||||
): Promise<string[]> {
|
||||
const files: string[] = []
|
||||
await forEachSessionFile(dirPath, agent, issues, options, async (path) => {
|
||||
files.push(path)
|
||||
})
|
||||
return files
|
||||
}
|
||||
|
||||
/** Streams matches to `onFile` so a bounded consumer never retains the whole tree. */
|
||||
export async function forEachSessionFile(
|
||||
dirPath: string,
|
||||
agent: AiVaultAgent,
|
||||
issues: AiVaultScanIssue[],
|
||||
options: SessionFileWalkOptions,
|
||||
onFile: (path: string) => Promise<void>,
|
||||
depth = 0
|
||||
): Promise<void> {
|
||||
options.signal?.throwIfAborted()
|
||||
let entries
|
||||
try {
|
||||
@@ -114,10 +139,9 @@ export async function walkSessionFiles(
|
||||
if (error instanceof WslTranscriptFsError) {
|
||||
throw error
|
||||
}
|
||||
return []
|
||||
return
|
||||
}
|
||||
|
||||
const files: string[] = []
|
||||
for (const entry of entries) {
|
||||
options.signal?.throwIfAborted()
|
||||
const fullPath = join(dirPath, entry.name)
|
||||
@@ -125,7 +149,7 @@ export async function walkSessionFiles(
|
||||
// Skip whole subtrees an agent never wants (e.g. subagent transcripts),
|
||||
// avoiding the readdir cost of descending into them.
|
||||
if (options.directoryPredicate?.(entry.name, depth) ?? true) {
|
||||
files.push(...(await walkSessionFiles(fullPath, agent, issues, options, depth + 1)))
|
||||
await forEachSessionFile(fullPath, agent, issues, options, onFile, depth + 1)
|
||||
}
|
||||
continue
|
||||
}
|
||||
@@ -134,12 +158,7 @@ export async function walkSessionFiles(
|
||||
options.extensions.has(extname(entry.name).toLowerCase()) &&
|
||||
(options.filePredicate?.(fullPath) ?? true)
|
||||
) {
|
||||
if (options.onFile) {
|
||||
await options.onFile(fullPath)
|
||||
} else {
|
||||
files.push(fullPath)
|
||||
}
|
||||
await onFile(fullPath)
|
||||
}
|
||||
}
|
||||
return files
|
||||
}
|
||||
|
||||
@@ -62,10 +62,7 @@ export async function consumeCompleteJsonlLines(args: {
|
||||
}
|
||||
newlineIndex = data.indexOf(NEWLINE_BYTE, lineStart)
|
||||
}
|
||||
const checkpoint = checkpointSessionSearchCapture()
|
||||
if (checkpoint) {
|
||||
await checkpoint
|
||||
}
|
||||
await checkpointSessionSearchCapture()
|
||||
consumedThrough += lineStart
|
||||
if (stopped) {
|
||||
remainderParts = []
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation'
|
||||
import type { OpenCodeSqliteWorkerRequest } from './session-scanner-opencode-sqlite-worker-protocol'
|
||||
import { getSessionSearchCaptureSignal } from './session-search-capture'
|
||||
import type { OpenCodeCaptureConsumer } from './session-search-opencode-capture-channel'
|
||||
|
||||
/** One request the client has accepted: queued or active, with its own deadline. */
|
||||
export type OpenCodePendingCall = {
|
||||
request: OpenCodeSqliteWorkerRequest
|
||||
timeoutMs: number
|
||||
resolve: (value: unknown) => void
|
||||
reject: (error: Error) => void
|
||||
timer: NodeJS.Timeout | null
|
||||
capture?: OpenCodeCaptureConsumer
|
||||
}
|
||||
|
||||
/**
|
||||
* The request owns its abort listener until either queued or active work settles.
|
||||
* Applies to every request kind, not just capturing parses: a backfill abort
|
||||
* cancels the `list` legs it queued too.
|
||||
*/
|
||||
export function bindOpenCodeRequestCancellation(
|
||||
resolve: (value: unknown) => void,
|
||||
reject: (error: Error) => void,
|
||||
cancel: () => void
|
||||
): { resolve: typeof resolve; reject: typeof reject } {
|
||||
const signal = getSessionSearchCaptureSignal()
|
||||
throwIfAiVaultScanCancelled(signal)
|
||||
signal?.addEventListener('abort', cancel, { once: true })
|
||||
const cleanup = (): void => signal?.removeEventListener('abort', cancel)
|
||||
return {
|
||||
resolve: (value) => {
|
||||
cleanup()
|
||||
resolve(value)
|
||||
},
|
||||
reject: (error) => {
|
||||
cleanup()
|
||||
reject(error)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { columnExists, tableExists } from '../opencode-usage/schema-helpers'
|
||||
|
||||
// Why: OpenCode's schema has moved more than once, so every read probes for the
|
||||
// columns it names. These are the two shapes the session parser depends on;
|
||||
// keeping them here stops each reader from inventing its own partial gate.
|
||||
|
||||
/** Enough of `message` to count a session's turns. */
|
||||
export function canCountOpenCodeMessages(db: SyncDatabase): boolean {
|
||||
return (
|
||||
tableExists(db, 'message') &&
|
||||
columnExists(db, 'message', 'session_id') &&
|
||||
columnExists(db, 'message', 'data')
|
||||
)
|
||||
}
|
||||
|
||||
/** Enough of `message`×`part` to read a session's parts in turn order. */
|
||||
export function canReadOpenCodeMessageParts(db: SyncDatabase): boolean {
|
||||
return (
|
||||
canCountOpenCodeMessages(db) &&
|
||||
columnExists(db, 'message', 'id') &&
|
||||
// Every parts read orders by it; unprobed, a schema without it throws mid-read.
|
||||
columnExists(db, 'message', 'time_created') &&
|
||||
tableExists(db, 'part') &&
|
||||
columnExists(db, 'part', 'message_id') &&
|
||||
columnExists(db, 'part', 'time_created') &&
|
||||
columnExists(db, 'part', 'data')
|
||||
)
|
||||
}
|
||||
@@ -1,12 +1,12 @@
|
||||
import type { Worker } from 'node:worker_threads'
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
IDLE_TEARDOWN_MS,
|
||||
LIST_TIMEOUT_MS,
|
||||
MAX_CONSECUTIVE_DEATHS,
|
||||
OpenCodeSqliteWorkerClient,
|
||||
PARSE_TIMEOUT_MS
|
||||
} from './session-scanner-opencode-sqlite-worker-client'
|
||||
import { IDLE_TEARDOWN_MS } from './session-scanner-opencode-worker-host'
|
||||
import type {
|
||||
OpenCodeSqliteParseValue,
|
||||
OpenCodeSqliteParentMessage,
|
||||
@@ -14,9 +14,9 @@ import type {
|
||||
} from './session-scanner-opencode-sqlite-worker-protocol'
|
||||
import {
|
||||
isSessionSearchCaptureActive,
|
||||
withSessionSearchCapture,
|
||||
withStreamingSessionSearchCapture
|
||||
} from './session-search-capture'
|
||||
import type { SessionSearchCapturedMessage } from './session-search-capture'
|
||||
import type { AiVaultScanIssue, AiVaultSession } from '../../shared/ai-vault-types'
|
||||
|
||||
// A worker_threads stand-in the tests drive directly: it records posted requests
|
||||
@@ -99,12 +99,12 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
// A response for a different id must not settle the active call.
|
||||
worker!.emit('message', {
|
||||
id: 999,
|
||||
ok: true,
|
||||
kind: 'result',
|
||||
value: null
|
||||
} satisfies OpenCodeSqliteWorkerResponse)
|
||||
worker!.emit('message', {
|
||||
id: worker!.lastId(),
|
||||
ok: true,
|
||||
kind: 'result',
|
||||
value: { session: { sessionId: 'a' } }
|
||||
} satisfies OpenCodeSqliteWorkerResponse)
|
||||
|
||||
@@ -123,12 +123,20 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
expect(worker.postedRequests).toHaveLength(1)
|
||||
expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', sessionId: 'a' })
|
||||
|
||||
worker.emit('message', { id: worker.postedRequests[0]!.id, ok: true, value: parseValue('A') })
|
||||
worker.emit('message', {
|
||||
id: worker.postedRequests[0]!.id,
|
||||
kind: 'result',
|
||||
value: parseValue('A')
|
||||
})
|
||||
await first
|
||||
|
||||
expect(worker.postedRequests).toHaveLength(2)
|
||||
expect(worker.postedRequests[1]).toMatchObject({ kind: 'parse', sessionId: 'b' })
|
||||
worker.emit('message', { id: worker.postedRequests[1]!.id, ok: true, value: parseValue('B') })
|
||||
worker.emit('message', {
|
||||
id: worker.postedRequests[1]!.id,
|
||||
kind: 'result',
|
||||
value: parseValue('B')
|
||||
})
|
||||
await expect(second).resolves.toBe('B')
|
||||
// The worker is reused across serial calls (one persistent worker).
|
||||
expect(workers).toHaveLength(1)
|
||||
@@ -157,7 +165,7 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
const respawned = workers[1]!
|
||||
expect(respawned.postedRequests).toHaveLength(1)
|
||||
expect(respawned.postedRequests[0]).toMatchObject({ sessionId: 'b' })
|
||||
respawned.emit('message', { id: respawned.lastId(), ok: true, value: parseValue('B') })
|
||||
respawned.emit('message', { id: respawned.lastId(), kind: 'result', value: parseValue('B') })
|
||||
await expect(queued).resolves.toBe('B')
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
@@ -179,7 +187,7 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
// Exactly one respawn; the queued call drains on the new worker.
|
||||
expect(workers).toHaveLength(2)
|
||||
const respawned = workers[1]!
|
||||
respawned.emit('message', { id: respawned.lastId(), ok: true, value: parseValue('B') })
|
||||
respawned.emit('message', { id: respawned.lastId(), kind: 'result', value: parseValue('B') })
|
||||
await expect(queued).resolves.toBe('B')
|
||||
})
|
||||
|
||||
@@ -285,7 +293,7 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
expect(worker).toBeDefined()
|
||||
worker!.emit('message', {
|
||||
id: worker!.lastId(),
|
||||
ok: true,
|
||||
kind: 'result',
|
||||
value: { candidates: [], issues: [] }
|
||||
} satisfies OpenCodeSqliteWorkerResponse)
|
||||
await expect(secondPromise).resolves.toEqual([])
|
||||
@@ -297,13 +305,21 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
|
||||
const first = client.parse({ dbPath: '/db#a', sessionId: 'a', platform: 'darwin' })
|
||||
workers[0]!.emit('message', { id: workers[0]!.lastId(), ok: true, value: parseValue('A') })
|
||||
workers[0]!.emit('message', {
|
||||
id: workers[0]!.lastId(),
|
||||
kind: 'result',
|
||||
value: parseValue('A')
|
||||
})
|
||||
await expect(first).resolves.toBe('A')
|
||||
workers[0]!.emit('exit', 0)
|
||||
|
||||
const second = client.parse({ dbPath: '/db#b', sessionId: 'b', platform: 'darwin' })
|
||||
expect(workers).toHaveLength(2)
|
||||
workers[1]!.emit('message', { id: workers[1]!.lastId(), ok: true, value: parseValue('B') })
|
||||
workers[1]!.emit('message', {
|
||||
id: workers[1]!.lastId(),
|
||||
kind: 'result',
|
||||
value: parseValue('B')
|
||||
})
|
||||
await expect(second).resolves.toBe('B')
|
||||
})
|
||||
|
||||
@@ -317,14 +333,22 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
})
|
||||
|
||||
const first = client.parse({ dbPath: '/db#a', sessionId: 'a', platform: 'darwin' })
|
||||
workers[0]!.emit('message', { id: workers[0]!.lastId(), ok: true, value: parseValue('A') })
|
||||
workers[0]!.emit('message', {
|
||||
id: workers[0]!.lastId(),
|
||||
kind: 'result',
|
||||
value: parseValue('A')
|
||||
})
|
||||
await expect(first).resolves.toBe('A')
|
||||
await vi.advanceTimersByTimeAsync(IDLE_TEARDOWN_MS)
|
||||
expect(workers[0]!.terminated).toBe(true)
|
||||
|
||||
const second = client.parse({ dbPath: '/db#b', sessionId: 'b', platform: 'darwin' })
|
||||
expect(workers).toHaveLength(2)
|
||||
workers[1]!.emit('message', { id: workers[1]!.lastId(), ok: true, value: parseValue('B') })
|
||||
workers[1]!.emit('message', {
|
||||
id: workers[1]!.lastId(),
|
||||
kind: 'result',
|
||||
value: parseValue('B')
|
||||
})
|
||||
await expect(second).resolves.toBe('B')
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
@@ -364,7 +388,7 @@ describe('OpenCodeSqliteWorkerClient', () => {
|
||||
for (let i = 0; i < 3; i++) {
|
||||
const promise = client.parse({ dbPath: `/db#${i}`, sessionId: `s${i}`, platform: 'darwin' })
|
||||
const worker = workers[0]!
|
||||
worker.emit('message', { id: worker.lastId(), ok: true, value: parseValue(`v${i}`) })
|
||||
worker.emit('message', { id: worker.lastId(), kind: 'result', value: parseValue(`v${i}`) })
|
||||
await expect(promise).resolves.toBe(`v${i}`)
|
||||
}
|
||||
expect(workers).toHaveLength(1)
|
||||
@@ -376,30 +400,34 @@ describe('OpenCodeSqliteWorkerClient search capture', () => {
|
||||
const workers: FakeWorker[] = []
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
|
||||
const captured = await withSessionSearchCapture(async () => {
|
||||
const parsePromise = client.parse({ dbPath: '/db', sessionId: 'a', platform: 'darwin' })
|
||||
const worker = workers[0]!
|
||||
await vi.waitFor(() => expect(worker.postedRequests).toHaveLength(1))
|
||||
expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', capture: true })
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
ok: true,
|
||||
captureBatch: 1,
|
||||
value: [{ role: 'user', text: 'ballast tanks', timestamp: null }]
|
||||
} satisfies OpenCodeSqliteWorkerResponse)
|
||||
await vi.waitFor(() =>
|
||||
expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 })
|
||||
)
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
ok: true,
|
||||
value: { session: { sessionId: 'a' } }
|
||||
})
|
||||
return parsePromise
|
||||
})
|
||||
const messages: SessionSearchCapturedMessage[] = []
|
||||
const value = await withStreamingSessionSearchCapture(
|
||||
{ push: (message) => messages.push(message), checkpoint: async () => {} },
|
||||
async () => {
|
||||
const parsePromise = client.parse({ dbPath: '/db', sessionId: 'a', platform: 'darwin' })
|
||||
const worker = workers[0]!
|
||||
await vi.waitFor(() => expect(worker.postedRequests).toHaveLength(1))
|
||||
expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', capture: true })
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
kind: 'batch',
|
||||
batch: 1,
|
||||
messages: [{ role: 'user', text: 'ballast tanks', timestamp: null }]
|
||||
} satisfies OpenCodeSqliteWorkerResponse)
|
||||
await vi.waitFor(() =>
|
||||
expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 })
|
||||
)
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
kind: 'result',
|
||||
value: { session: { sessionId: 'a' } }
|
||||
})
|
||||
return parsePromise
|
||||
}
|
||||
)
|
||||
|
||||
expect(captured.value).toEqual({ sessionId: 'a' })
|
||||
expect(captured.messages).toEqual([{ role: 'user', text: 'ballast tanks', timestamp: null }])
|
||||
expect(value).toEqual({ sessionId: 'a' })
|
||||
expect(messages).toEqual([{ role: 'user', text: 'ballast tanks', timestamp: null }])
|
||||
})
|
||||
|
||||
it('does not ask for capture outside a capture scope', async () => {
|
||||
@@ -412,7 +440,7 @@ describe('OpenCodeSqliteWorkerClient search capture', () => {
|
||||
expect(worker.postedRequests[0]).toMatchObject({ kind: 'parse', capture: false })
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
ok: true,
|
||||
kind: 'result',
|
||||
value: { session: null }
|
||||
} satisfies OpenCodeSqliteWorkerResponse)
|
||||
|
||||
@@ -435,9 +463,9 @@ it('acknowledges capture only after the caller channel drains, including across
|
||||
const worker = workers[0]!
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
ok: true,
|
||||
captureBatch: 1,
|
||||
value: [{ role: 'user', text: 'late marker', timestamp: null }]
|
||||
kind: 'batch',
|
||||
batch: 1,
|
||||
messages: [{ role: 'user', text: 'late marker', timestamp: null }]
|
||||
})
|
||||
await Promise.resolve()
|
||||
expect(push).toHaveBeenCalledOnce()
|
||||
@@ -447,14 +475,16 @@ it('acknowledges capture only after the caller channel drains, including across
|
||||
await vi.waitFor(() =>
|
||||
expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 })
|
||||
)
|
||||
worker.emit('message', { id: worker.lastId(), ok: true, value: { session: { sessionId: 'a' } } })
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
kind: 'result',
|
||||
value: { session: { sessionId: 'a' } }
|
||||
})
|
||||
expect(await parsed).toEqual({ sessionId: 'a' })
|
||||
})
|
||||
|
||||
it('rejects the parse and retires its worker when the capture consumer fails', async () => {
|
||||
const workers: FakeWorker[] = []
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
const parsed = withStreamingSessionSearchCapture(
|
||||
function failingCaptureParse(client: OpenCodeSqliteWorkerClient): Promise<unknown> {
|
||||
return withStreamingSessionSearchCapture(
|
||||
{
|
||||
push() {},
|
||||
checkpoint: async () => {
|
||||
@@ -463,37 +493,110 @@ it('rejects the parse and retires its worker when the capture consumer fails', a
|
||||
},
|
||||
() => client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform })
|
||||
)
|
||||
}
|
||||
|
||||
it('rejects the parse and retires its worker when the capture consumer fails', async () => {
|
||||
const workers: FakeWorker[] = []
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
const parsed = failingCaptureParse(client)
|
||||
const rejected = expect(parsed).rejects.toThrow('capture stopped')
|
||||
workers[0]!.emit('message', { id: workers[0]!.lastId(), ok: true, captureBatch: 1, value: [] })
|
||||
workers[0]!.emit('message', { id: workers[0]!.lastId(), kind: 'batch', batch: 1, messages: [] })
|
||||
await rejected
|
||||
// The worker is parked on an ack that will never arrive, so it has to go.
|
||||
expect(workers[0]!.terminated).toBe(true)
|
||||
expect(workers[0]!.postedRequests).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('suspends the producer timeout during backpressure and restores it after acknowledgement', async () => {
|
||||
it('does not count a failing index write as a worker death', async () => {
|
||||
const workers: FakeWorker[] = []
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
|
||||
// One burst: enough capturing parses to hit the respawn cap, plus an unrelated
|
||||
// list for another database queued behind them. The death counter only resets
|
||||
// from full idle, so nothing here hides an increment.
|
||||
const parses = withStreamingSessionSearchCapture(
|
||||
{
|
||||
push() {},
|
||||
checkpoint: async () => {
|
||||
throw new Error('capture stopped')
|
||||
}
|
||||
},
|
||||
async () =>
|
||||
Array.from({ length: MAX_CONSECUTIVE_DEATHS }, (_, i) =>
|
||||
client.parse({ dbPath: `/db#${i}`, sessionId: `s${i}`, platform: process.platform })
|
||||
)
|
||||
)
|
||||
const issues: AiVaultScanIssue[] = []
|
||||
const list = client.list({ dbPaths: ['/db#other'], limit: 10, issues })
|
||||
|
||||
for (const parse of await parses) {
|
||||
const rejected = expect(parse).rejects.toThrow('capture stopped')
|
||||
const worker = workers.at(-1)!
|
||||
worker.emit('message', { id: worker.lastId(), kind: 'batch', batch: 1, messages: [] })
|
||||
await rejected
|
||||
}
|
||||
|
||||
// A healthy worker parked on an unreachable ack is not a crash, so the queued
|
||||
// list must still be dispatched rather than drained as a crash loop.
|
||||
const worker = workers.at(-1)!
|
||||
expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'list' })
|
||||
worker.emit('message', {
|
||||
id: worker.lastId(),
|
||||
kind: 'result',
|
||||
value: { candidates: [], issues: [] }
|
||||
})
|
||||
await expect(list).resolves.toEqual([])
|
||||
expect(issues).toEqual([])
|
||||
})
|
||||
|
||||
it('caps a single backpressure stall at the parse deadline instead of waiting forever', async () => {
|
||||
vi.useFakeTimers()
|
||||
try {
|
||||
const workers: FakeWorker[] = []
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
let release!: () => void
|
||||
const drained = new Promise<void>((resolve) => {
|
||||
release = resolve
|
||||
})
|
||||
const parsed = withStreamingSessionSearchCapture({ push() {}, checkpoint: () => drained }, () =>
|
||||
client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform })
|
||||
// A consumer that never resolves stands in for a wedged index writer.
|
||||
const parsed = withStreamingSessionSearchCapture(
|
||||
{ push() {}, checkpoint: () => new Promise<void>(() => {}) },
|
||||
() => client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform })
|
||||
)
|
||||
const failure = expect(parsed).rejects.toThrow('timed out')
|
||||
const worker = workers[0]!
|
||||
worker.emit('message', { id: worker.lastId(), ok: true, captureBatch: 1, value: [] })
|
||||
await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS + 1)
|
||||
worker.emit('message', { id: worker.lastId(), kind: 'batch', batch: 1, messages: [] })
|
||||
|
||||
// The batch restarts the deadline rather than removing it.
|
||||
await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS - 1)
|
||||
expect(worker.terminated).toBe(false)
|
||||
release()
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 1 })
|
||||
await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS)
|
||||
await vi.advanceTimersByTimeAsync(2)
|
||||
await failure
|
||||
expect(worker.terminated).toBe(true)
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
}
|
||||
})
|
||||
|
||||
it('lets total production run past the deadline as long as batches keep arriving', async () => {
|
||||
vi.useFakeTimers()
|
||||
try {
|
||||
const workers: FakeWorker[] = []
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: makeFactory(workers), log() {} })
|
||||
const parsed = withStreamingSessionSearchCapture(
|
||||
{ push() {}, checkpoint: async () => {} },
|
||||
() => client.parse({ dbPath: '/db', sessionId: 'a', platform: process.platform })
|
||||
)
|
||||
const worker = workers[0]!
|
||||
|
||||
// Four batches, each landing just under the deadline: cumulative time is far
|
||||
// past PARSE_TIMEOUT_MS, but no single gap is, so the parse must survive.
|
||||
for (let batch = 1; batch <= 4; batch++) {
|
||||
worker.emit('message', { id: worker.lastId(), kind: 'batch', batch, messages: [] })
|
||||
await vi.advanceTimersByTimeAsync(PARSE_TIMEOUT_MS - 1)
|
||||
expect(worker.terminated).toBe(false)
|
||||
}
|
||||
expect(worker.postedRequests.at(-1)).toMatchObject({ kind: 'captureAck', batch: 4 })
|
||||
|
||||
worker.emit('message', { id: worker.lastId(), kind: 'result', value: parseValue('A') })
|
||||
await expect(parsed).resolves.toBe('A')
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
}
|
||||
})
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
import type { Worker } from 'node:worker_threads'
|
||||
import type { AiVaultScanIssue, AiVaultSession } from '../../shared/ai-vault-types'
|
||||
import type {
|
||||
OpenCodeSqliteListValue,
|
||||
@@ -9,11 +8,17 @@ import type {
|
||||
import { createAiVaultScanCancelledError } from './ai-vault-scan-cancellation'
|
||||
import { isSessionSearchCaptureActive } from './session-search-capture'
|
||||
import {
|
||||
bindOpenCodeCaptureConsumer,
|
||||
bindOpenCodeCaptureCancellation,
|
||||
receiveOpenCodeCaptureBatch,
|
||||
bindOpenCodeRequestCancellation,
|
||||
type OpenCodePendingCall as PendingCall
|
||||
} from './session-search-opencode-worker-receiver'
|
||||
} from './session-scanner-opencode-pending-request'
|
||||
import {
|
||||
bindOpenCodeCaptureConsumer,
|
||||
receiveOpenCodeCaptureBatch
|
||||
} from './session-search-opencode-capture-channel'
|
||||
import {
|
||||
OpenCodeSqliteWorkerHost,
|
||||
type WorkerFactory
|
||||
} from './session-scanner-opencode-worker-host'
|
||||
import type { SessionFileCandidate } from './session-scanner-types'
|
||||
import { errorMessage } from './session-scanner-values'
|
||||
|
||||
@@ -25,15 +30,12 @@ import { errorMessage } from './session-scanner-values'
|
||||
|
||||
export const LIST_TIMEOUT_MS = 30_000
|
||||
export const PARSE_TIMEOUT_MS = 15_000
|
||||
export const IDLE_TEARDOWN_MS = 30_000
|
||||
// After this many consecutive worker deaths, fail the remaining queued calls to
|
||||
// scan issues instead of respawning so a DB that reliably kills the worker can't
|
||||
// spin a crash loop. Reset on any successful response, after draining, and when a
|
||||
// fresh scan burst starts from idle (so the cap is per-scan, not process-wide).
|
||||
export const MAX_CONSECUTIVE_DEATHS = 3
|
||||
|
||||
export type WorkerFactory = () => Worker
|
||||
|
||||
// Distinguishes "no worker available at all" from a timeout or crash so callers
|
||||
// can surface a precise issue while keeping synchronous SQLite off the main thread.
|
||||
class OpenCodeSqliteWorkerUnavailableError extends Error {}
|
||||
@@ -46,20 +48,21 @@ class OpenCodeSqliteWorkerUnavailableError extends Error {}
|
||||
* no worker can be spawned rather than moving SQLite work onto the main thread.
|
||||
*/
|
||||
export class OpenCodeSqliteWorkerClient {
|
||||
private worker: Worker | null = null
|
||||
private active: PendingCall | null = null
|
||||
private queue: PendingCall[] = []
|
||||
private idleTimer: NodeJS.Timeout | null = null
|
||||
private consecutiveDeaths = 0
|
||||
private nextId = 1
|
||||
private loggedWorkerUnavailable = false
|
||||
private cleanupWorkerListeners: (() => void) | null = null
|
||||
private readonly workerFactory: WorkerFactory
|
||||
private readonly log: (message: string) => void
|
||||
private readonly host: OpenCodeSqliteWorkerHost
|
||||
|
||||
constructor(options: { workerFactory: WorkerFactory; log?: (message: string) => void }) {
|
||||
this.workerFactory = options.workerFactory
|
||||
this.log = options.log ?? ((message) => console.warn(message))
|
||||
this.host = new OpenCodeSqliteWorkerHost({
|
||||
factory: options.workerFactory,
|
||||
log: options.log ?? ((message) => console.warn(message)),
|
||||
onMessage: (response) => this.onMessage(response),
|
||||
onError: (error) => this.onWorkerFault(error),
|
||||
onExit: (code) => this.onWorkerExit(code),
|
||||
isIdle: () => !this.active && this.queue.length === 0
|
||||
})
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -160,7 +163,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
const call: PendingCall = {
|
||||
request: { ...request, id } as PendingCall['request'],
|
||||
timeoutMs,
|
||||
...bindOpenCodeCaptureCancellation(resolve, reject, () => this.cancel(call)),
|
||||
...bindOpenCodeRequestCancellation(resolve, reject, () => this.cancel(call)),
|
||||
timer: null,
|
||||
capture
|
||||
}
|
||||
@@ -171,7 +174,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
|
||||
private cancel(call: PendingCall): void {
|
||||
if (this.active === call) {
|
||||
this.destroyWorker()
|
||||
this.host.destroy()
|
||||
}
|
||||
this.queue = this.queue.filter((pending) => pending !== call)
|
||||
this.settle(call, () => call.reject(createAiVaultScanCancelledError()))
|
||||
@@ -182,7 +185,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
if (this.active || this.queue.length === 0) {
|
||||
return
|
||||
}
|
||||
const worker = this.ensureWorker()
|
||||
const worker = this.host.ensure()
|
||||
if (!worker) {
|
||||
this.failQueuedAsUnavailable()
|
||||
return
|
||||
@@ -192,7 +195,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
return
|
||||
}
|
||||
this.active = call
|
||||
this.clearIdleTimer()
|
||||
this.host.clearIdleTimer()
|
||||
// Timeout clock starts at dispatch (not enqueue): a batch may enqueue up to
|
||||
// 8 parses at once, and a queue-inclusive timeout would fire falsely.
|
||||
call.timer = setTimeout(() => this.onTimeout(call), call.timeoutMs)
|
||||
@@ -200,61 +203,38 @@ export class OpenCodeSqliteWorkerClient {
|
||||
worker.postMessage(call.request)
|
||||
}
|
||||
|
||||
private ensureWorker(): Worker | null {
|
||||
if (this.worker) {
|
||||
return this.worker
|
||||
}
|
||||
try {
|
||||
const worker = this.workerFactory()
|
||||
const onMessage = (response: OpenCodeSqliteWorkerResponse): void => this.onMessage(response)
|
||||
const onError = (error: Error): void => this.onWorkerFault(error)
|
||||
const onExit = (code: number): void => this.onWorkerExit(code)
|
||||
worker.on('message', onMessage)
|
||||
worker.on('error', onError)
|
||||
worker.on('exit', onExit)
|
||||
this.cleanupWorkerListeners = () => {
|
||||
worker.off('message', onMessage)
|
||||
worker.off('error', onError)
|
||||
worker.off('exit', onExit)
|
||||
}
|
||||
// Never keep the app alive for a scan worker.
|
||||
worker.unref?.()
|
||||
this.worker = worker
|
||||
return worker
|
||||
} catch (err) {
|
||||
// Why (#8864): never fall back to synchronous SQLite reads here; a missing
|
||||
// bundle or resource-exhausted spawn must omit OpenCode history rather than
|
||||
// reintroduce the main-process hang this worker boundary prevents.
|
||||
if (!this.loggedWorkerUnavailable) {
|
||||
this.loggedWorkerUnavailable = true
|
||||
this.log(`OpenCode SQLite worker unavailable; skipping its history. ${errorMessage(err)}`)
|
||||
}
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
private onMessage(response: OpenCodeSqliteWorkerResponse): void {
|
||||
const call = this.active
|
||||
if (!call || call.request.id !== response.id) {
|
||||
return
|
||||
}
|
||||
if (response.ok && response.captureBatch !== undefined) {
|
||||
if (response.kind === 'batch') {
|
||||
receiveOpenCodeCaptureBatch({
|
||||
call,
|
||||
response,
|
||||
worker: this.worker,
|
||||
batch: response,
|
||||
worker: this.host.current,
|
||||
isActive: () => this.active === call,
|
||||
onTimeout: () => this.onTimeout(call),
|
||||
onError: (error) => this.onWorkerFault(error)
|
||||
onProtocolViolation: (error) => this.onWorkerFault(error),
|
||||
onConsumerError: (error) => this.onCaptureConsumerFailure(call, error)
|
||||
})
|
||||
return
|
||||
}
|
||||
this.consecutiveDeaths = 0
|
||||
if (response.ok) {
|
||||
this.settle(call, () => call.resolve(response.value))
|
||||
} else {
|
||||
this.settle(call, () => call.reject(new Error(response.error)))
|
||||
}
|
||||
this.settle(call, () =>
|
||||
response.kind === 'result'
|
||||
? call.resolve(response.value)
|
||||
: call.reject(new Error(response.error))
|
||||
)
|
||||
this.afterSettle()
|
||||
}
|
||||
|
||||
// The worker is healthy, only parked on an ack that will never arrive, so it
|
||||
// is retired without counting a death: a failing index write must not spend
|
||||
// the respawn budget that unrelated queued calls depend on.
|
||||
private onCaptureConsumerFailure(call: PendingCall, error: Error): void {
|
||||
this.host.destroy()
|
||||
this.settle(call, () => call.reject(error))
|
||||
this.afterSettle()
|
||||
}
|
||||
|
||||
@@ -269,7 +249,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
// A clean self-exit is not a death, but the stale handle must be dropped
|
||||
// or the next dispatch would post into the dead worker and stall to timeout.
|
||||
if (code === 0 && !this.active && this.queue.length === 0) {
|
||||
this.destroyWorker()
|
||||
this.host.destroy()
|
||||
return
|
||||
}
|
||||
this.onWorkerFault(new Error(`OpenCode SQLite worker exited with code ${code}`))
|
||||
@@ -277,7 +257,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
|
||||
private onWorkerFault(error: Error): void {
|
||||
const failed = this.active
|
||||
this.destroyWorker()
|
||||
this.host.destroy()
|
||||
this.consecutiveDeaths++
|
||||
if (failed) {
|
||||
this.settle(failed, () => failed.reject(error))
|
||||
@@ -328,46 +308,7 @@ export class OpenCodeSqliteWorkerClient {
|
||||
if (this.queue.length > 0) {
|
||||
this.pump()
|
||||
} else {
|
||||
this.scheduleIdleTeardown()
|
||||
this.host.scheduleIdleTeardown()
|
||||
}
|
||||
}
|
||||
|
||||
private scheduleIdleTeardown(): void {
|
||||
this.clearIdleTimer()
|
||||
if (!this.worker) {
|
||||
return
|
||||
}
|
||||
this.idleTimer = setTimeout(() => this.teardownIfIdle(), IDLE_TEARDOWN_MS)
|
||||
this.idleTimer.unref?.()
|
||||
}
|
||||
|
||||
private teardownIfIdle(): void {
|
||||
this.idleTimer = null
|
||||
// Only tear down with nothing active AND nothing queued: a request arriving
|
||||
// as the timer fires must never be lost to a self-exiting worker.
|
||||
if (this.active || this.queue.length > 0) {
|
||||
return
|
||||
}
|
||||
this.destroyWorker()
|
||||
}
|
||||
|
||||
private clearIdleTimer(): void {
|
||||
if (this.idleTimer) {
|
||||
clearTimeout(this.idleTimer)
|
||||
this.idleTimer = null
|
||||
}
|
||||
}
|
||||
|
||||
private destroyWorker(): void {
|
||||
this.clearIdleTimer()
|
||||
const worker = this.worker
|
||||
this.worker = null
|
||||
if (!worker) {
|
||||
return
|
||||
}
|
||||
this.cleanupWorkerListeners?.()
|
||||
this.cleanupWorkerListeners = null
|
||||
worker.removeAllListeners()
|
||||
void worker.terminate().catch(() => undefined)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -32,9 +32,9 @@ vi.mock('node:worker_threads', () => ({
|
||||
},
|
||||
postMessage(response: OpenCodeSqliteWorkerResponse) {
|
||||
posted.push(response)
|
||||
if (acknowledge && response.ok && response.captureBatch !== undefined) {
|
||||
if (acknowledge && response.kind === 'batch') {
|
||||
queueMicrotask(() =>
|
||||
handler?.({ id: response.id, kind: 'captureAck', batch: response.captureBatch! })
|
||||
handler?.({ id: response.id, kind: 'captureAck', batch: response.batch })
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -89,13 +89,14 @@ function createDbWithOneTurn(): string {
|
||||
|
||||
async function parseOnWorker(dbPath: string, capture: boolean): Promise<OpenCodeSqliteParseValue> {
|
||||
handler?.({ id: 1, kind: 'parse', dbPath, sessionId: SESSION_ID, platform: 'darwin', capture })
|
||||
await vi.waitFor(() =>
|
||||
expect(posted.some((reply) => !reply.ok || reply.captureBatch === undefined)).toBe(true)
|
||||
)
|
||||
await vi.waitFor(() => expect(posted.some((reply) => reply.kind !== 'batch')).toBe(true))
|
||||
const response = posted.at(-1)!
|
||||
if (!response.ok) {
|
||||
if (response.kind === 'error') {
|
||||
throw new Error(response.error)
|
||||
}
|
||||
if (response.kind !== 'result') {
|
||||
throw new Error('worker replied with a capture batch instead of a result')
|
||||
}
|
||||
return response.value as OpenCodeSqliteParseValue
|
||||
}
|
||||
|
||||
@@ -104,11 +105,7 @@ describe('OpenCode SQLite worker entry', () => {
|
||||
const value = await parseOnWorker(createDbWithOneTurn(), true)
|
||||
|
||||
expect(value.session?.sessionId).toBe(SESSION_ID)
|
||||
expect(
|
||||
posted
|
||||
.filter((reply) => reply.ok && reply.captureBatch !== undefined)
|
||||
.flatMap((reply) => (reply.ok ? reply.value : []))
|
||||
).toEqual([
|
||||
expect(posted.flatMap((reply) => (reply.kind === 'batch' ? reply.messages : []))).toEqual([
|
||||
{ role: 'user', text: 'recalibrate the ballast pump', timestamp: expect.any(String) }
|
||||
])
|
||||
})
|
||||
@@ -117,11 +114,7 @@ describe('OpenCode SQLite worker entry', () => {
|
||||
const value = await parseOnWorker(createDbWithOneTurn(), false)
|
||||
|
||||
expect(value.session?.sessionId).toBe(SESSION_ID)
|
||||
expect(
|
||||
posted
|
||||
.filter((reply) => reply.ok && reply.captureBatch !== undefined)
|
||||
.flatMap((reply) => (reply.ok ? reply.value : []))
|
||||
).toEqual([])
|
||||
expect(posted.flatMap((reply) => (reply.kind === 'batch' ? reply.messages : []))).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -153,17 +146,18 @@ it('waits for downstream acknowledgement between bounded batches without droppin
|
||||
})
|
||||
await vi.waitFor(() => expect(posted).toHaveLength(1))
|
||||
const first = posted[0]!
|
||||
expect(first).toMatchObject({ ok: true, captureBatch: 1 })
|
||||
expect(first).toMatchObject({ kind: 'batch', batch: 1 })
|
||||
await new Promise((resolve) => setTimeout(resolve, 20))
|
||||
expect(posted).toHaveLength(1)
|
||||
acknowledge = true
|
||||
handler?.({ id: 2, kind: 'captureAck', batch: 1 })
|
||||
await vi.waitFor(() =>
|
||||
expect(posted.at(-1)).toMatchObject({ ok: true, value: { session: { sessionId: SESSION_ID } } })
|
||||
)
|
||||
const batches = posted.flatMap((reply) =>
|
||||
reply.ok && reply.captureBatch !== undefined ? [reply.value as { text: string }[]] : []
|
||||
expect(posted.at(-1)).toMatchObject({
|
||||
kind: 'result',
|
||||
value: { session: { sessionId: SESSION_ID } }
|
||||
})
|
||||
)
|
||||
const batches = posted.flatMap((reply) => (reply.kind === 'batch' ? [reply.messages] : []))
|
||||
expect(batches.length).toBeGreaterThan(2)
|
||||
expect(batches.flat()).toHaveLength(count + 1)
|
||||
expect(batches.flat().at(-1)?.text).toBe(`${count - 1} ${text}`.trim())
|
||||
|
||||
@@ -33,11 +33,15 @@ async function handleRequest(
|
||||
limit: request.limit,
|
||||
issues
|
||||
})
|
||||
return { id: request.id, ok: true, value: { candidates, issues } }
|
||||
return { id: request.id, kind: 'result', value: { candidates, issues } }
|
||||
}
|
||||
return { id: request.id, ok: true, value: await parseSession(request) }
|
||||
return { id: request.id, kind: 'result', value: await parseSession(request) }
|
||||
} catch (err) {
|
||||
return { id: request.id, ok: false, error: err instanceof Error ? err.message : String(err) }
|
||||
return {
|
||||
id: request.id,
|
||||
kind: 'error',
|
||||
error: err instanceof Error ? err.message : String(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -78,7 +82,7 @@ port.on('message', (request: OpenCodeSqliteParentMessage) => {
|
||||
// waiting out its timeout; fail that request fast instead.
|
||||
port.postMessage({
|
||||
id: request.id,
|
||||
ok: false,
|
||||
kind: 'error',
|
||||
error: 'OpenCode SQLite worker result could not be serialized.'
|
||||
})
|
||||
}
|
||||
|
||||
@@ -39,18 +39,27 @@ export type OpenCodeSqliteParseValue = {
|
||||
session: AiVaultSession | null
|
||||
}
|
||||
|
||||
// One acknowledged slice of a parse's index rows, sent before that parse's
|
||||
// result. `batch` numbers them so a late ack cannot release the wrong one.
|
||||
export type OpenCodeSqliteCaptureBatch = {
|
||||
id: number
|
||||
kind: 'batch'
|
||||
batch: number
|
||||
messages: SessionSearchCapturedMessage[]
|
||||
}
|
||||
|
||||
export type OpenCodeSqliteWorkerResult = { id: number; kind: 'result'; value: unknown }
|
||||
export type OpenCodeSqliteWorkerFailure = { id: number; kind: 'error'; error: string }
|
||||
|
||||
// Tagged rather than a boolean plus an optional field: the tag is what lets the
|
||||
// client narrow to the batch shape without casting its payload.
|
||||
export type OpenCodeSqliteWorkerResponse =
|
||||
| { id: number; ok: true; value: unknown; captureBatch?: number }
|
||||
| { id: number; ok: false; error: string }
|
||||
| OpenCodeSqliteCaptureBatch
|
||||
| OpenCodeSqliteWorkerResult
|
||||
| OpenCodeSqliteWorkerFailure
|
||||
|
||||
export type OpenCodeSqliteCaptureAck = { id: number; kind: 'captureAck'; batch: number }
|
||||
export type OpenCodeSqliteParentMessage = OpenCodeSqliteWorkerRequest | OpenCodeSqliteCaptureAck
|
||||
export type OpenCodeSqliteCaptureBatch = {
|
||||
id: number
|
||||
ok: true
|
||||
captureBatch: number
|
||||
value: SessionSearchCapturedMessage[]
|
||||
}
|
||||
|
||||
export type OpenCodeSqliteRequestBody =
|
||||
| Omit<OpenCodeSqliteListRequest, 'id'>
|
||||
|
||||
@@ -443,4 +443,47 @@ describe('parseOpenCodeSqliteSession', () => {
|
||||
|
||||
expect(session!.firstUserPrompt).toBe('the real typed ask')
|
||||
})
|
||||
|
||||
it('still returns the session when a corrupt part blob breaks the preview read', async () => {
|
||||
const { db, path } = createTempDb()
|
||||
applyOpenCodeSqliteSchema(db)
|
||||
insertOpenCodeSession(db, {
|
||||
id: 'ses_badblob',
|
||||
title: 'Ballast planning',
|
||||
timeCreated: 1_777_634_000_000,
|
||||
timeUpdated: 1_777_634_900_000
|
||||
})
|
||||
insertOpenCodeMessage(db, {
|
||||
id: 'msg_1',
|
||||
sessionId: 'ses_badblob',
|
||||
role: 'user',
|
||||
timeCreated: 1_777_634_000_000
|
||||
})
|
||||
insertOpenCodePart(db, {
|
||||
id: 'part_ok',
|
||||
messageId: 'msg_1',
|
||||
sessionId: 'ses_badblob',
|
||||
timeCreated: 10,
|
||||
text: 'readable turn'
|
||||
})
|
||||
// Truncated JSON, so the preview query's json_extract raises rather than
|
||||
// returning NULL. One corrupt row must not cost the session its listing.
|
||||
db.prepare(
|
||||
`INSERT INTO part (id, message_id, session_id, time_created, time_updated, data)
|
||||
VALUES ('part_bad', 'msg_1', 'ses_badblob', 20, 20, '{"type":')`
|
||||
).run()
|
||||
db.close()
|
||||
|
||||
const session = await parseOpenCodeSqliteSession({
|
||||
dbPath: path,
|
||||
sessionId: 'ses_badblob',
|
||||
platform: 'darwin'
|
||||
})
|
||||
|
||||
expect(session).not.toBeNull()
|
||||
expect(session!.sessionId).toBe('ses_badblob')
|
||||
expect(session!.title).toBe('Ballast planning')
|
||||
// The preview degrades to empty rather than taking the session with it.
|
||||
expect(session!.previewMessages).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -12,6 +12,10 @@ import {
|
||||
shouldCaptureFullFirstUserPrompt
|
||||
} from './session-scanner-first-user-prompt'
|
||||
import { readOpenCodeDatabaseAsync } from './session-scanner-opencode-sqlite-open'
|
||||
import {
|
||||
canCountOpenCodeMessages,
|
||||
canReadOpenCodeMessageParts
|
||||
} from './session-scanner-opencode-sqlite-schema'
|
||||
import { normalizeTitleText } from './session-scanner-values'
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { columnExists, tableExists } from '../opencode-usage/schema-helpers'
|
||||
@@ -72,14 +76,6 @@ function sessionNumberColumnSelect(db: SyncDatabase, columnName: string): string
|
||||
return columnExists(db, 'session', columnName) ? `s.${columnName}` : '0'
|
||||
}
|
||||
|
||||
function canCountOpenCodeMessages(db: SyncDatabase): boolean {
|
||||
return (
|
||||
tableExists(db, 'message') &&
|
||||
columnExists(db, 'message', 'session_id') &&
|
||||
columnExists(db, 'message', 'data')
|
||||
)
|
||||
}
|
||||
|
||||
function buildSessionQuery(db: SyncDatabase): string {
|
||||
const messageCountSubquery = canCountOpenCodeMessages(db)
|
||||
? `(SELECT COUNT(*) FROM message m
|
||||
@@ -156,14 +152,7 @@ function extractPartText(partData: string): string | null {
|
||||
}
|
||||
|
||||
function readFirstUserPromptFromOpenCodeDb(db: SyncDatabase, sessionId: string): string | null {
|
||||
if (
|
||||
!canCountOpenCodeMessages(db) ||
|
||||
!tableExists(db, 'part') ||
|
||||
!columnExists(db, 'message', 'id') ||
|
||||
!columnExists(db, 'part', 'message_id') ||
|
||||
!columnExists(db, 'part', 'time_created') ||
|
||||
!columnExists(db, 'part', 'data')
|
||||
) {
|
||||
if (!canReadOpenCodeMessageParts(db)) {
|
||||
return null
|
||||
}
|
||||
|
||||
@@ -208,14 +197,7 @@ function readFirstUserPromptFromOpenCodeDb(db: SyncDatabase, sessionId: string):
|
||||
}
|
||||
|
||||
function buildPreviewQuery(db: SyncDatabase): string | null {
|
||||
if (
|
||||
!canCountOpenCodeMessages(db) ||
|
||||
!tableExists(db, 'part') ||
|
||||
!columnExists(db, 'message', 'id') ||
|
||||
!columnExists(db, 'part', 'message_id') ||
|
||||
!columnExists(db, 'part', 'time_created') ||
|
||||
!columnExists(db, 'part', 'data')
|
||||
) {
|
||||
if (!canReadOpenCodeMessageParts(db)) {
|
||||
return null
|
||||
}
|
||||
return `SELECT json_extract(m.data, '$.role') AS role,
|
||||
@@ -234,6 +216,29 @@ function buildPreviewQuery(db: SyncDatabase): string | null {
|
||||
LIMIT ?`
|
||||
}
|
||||
|
||||
/**
|
||||
* Run the preview query, or null when it cannot be read.
|
||||
*
|
||||
* `json_extract` raises on a malformed part blob, so an unguarded read here
|
||||
* would drop the whole session from the list over one corrupt row. Degrade to
|
||||
* "no preview" instead, matching every other read in this module.
|
||||
*/
|
||||
function readPreviewRows(
|
||||
db: SyncDatabase,
|
||||
previewSql: string,
|
||||
sessionId: string
|
||||
): PreviewRow[] | null {
|
||||
try {
|
||||
return db.prepare(previewSql).all(sessionId, OPENCODE_SQLITE_PREVIEW_LIMIT + 1) as PreviewRow[]
|
||||
} catch (error) {
|
||||
console.warn(
|
||||
'[ai-vault] opencode preview skipped',
|
||||
error instanceof Error ? error.name : 'ReadError'
|
||||
)
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a single OpenCode session from the SQLite database into an
|
||||
* `AiVaultSession`. Reads session metadata (title, cwd, model, tokens, cost)
|
||||
@@ -298,14 +303,11 @@ async function readSession(args: {
|
||||
updateTimeline(accumulator, row.time_updated)
|
||||
|
||||
const previewSql = buildPreviewQuery(db)
|
||||
if (previewSql) {
|
||||
await captureOpenCodeSession(db, sessionId)
|
||||
// Why: SQL already dropped anything older than the newest-N window, so the
|
||||
// accumulator never shifts and cannot detect the truncation itself. Ask for
|
||||
// one extra row so an exactly-full window is not mistaken for a trimmed one.
|
||||
const probedRows = db
|
||||
.prepare(previewSql)
|
||||
.all(sessionId, OPENCODE_SQLITE_PREVIEW_LIMIT + 1) as PreviewRow[]
|
||||
// Why: SQL already dropped anything older than the newest-N window, so the
|
||||
// accumulator never shifts and cannot detect the truncation itself. Ask for
|
||||
// one extra row so an exactly-full window is not mistaken for a trimmed one.
|
||||
const probedRows = previewSql ? readPreviewRows(db, previewSql, sessionId) : null
|
||||
if (probedRows) {
|
||||
if (probedRows.length > OPENCODE_SQLITE_PREVIEW_LIMIT) {
|
||||
accumulator.previewMessagesTruncated = true
|
||||
}
|
||||
@@ -340,6 +342,8 @@ async function readSession(args: {
|
||||
}
|
||||
}
|
||||
|
||||
await captureOpenCodeSession(db, sessionId)
|
||||
|
||||
// Why: list preview only joins the newest messages. On-demand copy needs the
|
||||
// session's earliest real user text part, not a later turn still in the window.
|
||||
if (shouldCaptureFullFirstUserPrompt()) {
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
import type { Worker } from 'node:worker_threads'
|
||||
import type { OpenCodeSqliteWorkerResponse } from './session-scanner-opencode-sqlite-worker-protocol'
|
||||
import { errorMessage } from './session-scanner-values'
|
||||
|
||||
export type WorkerFactory = () => Worker
|
||||
|
||||
export const IDLE_TEARDOWN_MS = 30_000
|
||||
|
||||
/**
|
||||
* Owns the lifetime of the one OpenCode SQLite worker thread: lazy spawn,
|
||||
* listener wiring, teardown, and idle expiry. It holds no request state, so
|
||||
* every decision about which call a message belongs to stays with the client.
|
||||
*/
|
||||
export class OpenCodeSqliteWorkerHost {
|
||||
private worker: Worker | null = null
|
||||
private idleTimer: NodeJS.Timeout | null = null
|
||||
private cleanupListeners: (() => void) | null = null
|
||||
private loggedUnavailable = false
|
||||
|
||||
constructor(
|
||||
private readonly options: {
|
||||
factory: WorkerFactory
|
||||
log: (message: string) => void
|
||||
onMessage: (response: OpenCodeSqliteWorkerResponse) => void
|
||||
onError: (error: Error) => void
|
||||
onExit: (code: number) => void
|
||||
/** Nothing active and nothing queued, checked again when the idle timer fires. */
|
||||
isIdle: () => boolean
|
||||
}
|
||||
) {}
|
||||
|
||||
get current(): Worker | null {
|
||||
return this.worker
|
||||
}
|
||||
|
||||
/** The live worker, spawning one if needed; null when no worker can be had. */
|
||||
ensure(): Worker | null {
|
||||
if (this.worker) {
|
||||
return this.worker
|
||||
}
|
||||
try {
|
||||
const worker = this.options.factory()
|
||||
const onMessage = (response: OpenCodeSqliteWorkerResponse): void =>
|
||||
this.options.onMessage(response)
|
||||
const onError = (error: Error): void => this.options.onError(error)
|
||||
const onExit = (code: number): void => this.options.onExit(code)
|
||||
worker.on('message', onMessage)
|
||||
worker.on('error', onError)
|
||||
worker.on('exit', onExit)
|
||||
this.cleanupListeners = () => {
|
||||
worker.off('message', onMessage)
|
||||
worker.off('error', onError)
|
||||
worker.off('exit', onExit)
|
||||
}
|
||||
// Never keep the app alive for a scan worker.
|
||||
worker.unref?.()
|
||||
this.worker = worker
|
||||
return worker
|
||||
} catch (err) {
|
||||
// Why (#8864): never fall back to synchronous SQLite reads here; a missing
|
||||
// bundle or resource-exhausted spawn must omit OpenCode history rather than
|
||||
// reintroduce the main-process hang this worker boundary prevents.
|
||||
if (!this.loggedUnavailable) {
|
||||
this.loggedUnavailable = true
|
||||
this.options.log(
|
||||
`OpenCode SQLite worker unavailable; skipping its history. ${errorMessage(err)}`
|
||||
)
|
||||
}
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
destroy(): void {
|
||||
this.clearIdleTimer()
|
||||
const worker = this.worker
|
||||
this.worker = null
|
||||
if (!worker) {
|
||||
return
|
||||
}
|
||||
this.cleanupListeners?.()
|
||||
this.cleanupListeners = null
|
||||
worker.removeAllListeners()
|
||||
void worker.terminate().catch(() => undefined)
|
||||
}
|
||||
|
||||
scheduleIdleTeardown(): void {
|
||||
this.clearIdleTimer()
|
||||
if (!this.worker) {
|
||||
return
|
||||
}
|
||||
this.idleTimer = setTimeout(() => {
|
||||
this.idleTimer = null
|
||||
// Re-checked here: a request arriving as the timer fires must never be
|
||||
// lost to a self-exiting worker.
|
||||
if (this.options.isIdle()) {
|
||||
this.destroy()
|
||||
}
|
||||
}, IDLE_TEARDOWN_MS)
|
||||
this.idleTimer.unref?.()
|
||||
}
|
||||
|
||||
clearIdleTimer(): void {
|
||||
if (this.idleTimer) {
|
||||
clearTimeout(this.idleTimer)
|
||||
this.idleTimer = null
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -177,8 +177,13 @@ async function parseCachedInLane(
|
||||
// Codex titles come from session_index.jsonl, which mtime+size can't see.
|
||||
// Remote counterpart: remote-session-scanner.ts's reusedCodexTitleRefresh.
|
||||
if (entry.session && candidate.agent === 'codex') {
|
||||
// Why: the refresh returns the same reference when nothing changed, and a
|
||||
// list scan runs every ~5 s; writing regardless is one UPDATE per session.
|
||||
const previous = entry.session
|
||||
entry.session = await refreshCachedCodexMetadata(candidate, entry.session)
|
||||
sink?.updateMetadata?.(candidate, entry.session)
|
||||
if (entry.session !== previous) {
|
||||
sink?.updateMetadata?.(candidate, entry.session)
|
||||
}
|
||||
}
|
||||
storeSessionParseCacheEntry(file.path, entry)
|
||||
return entry.session
|
||||
@@ -318,6 +323,9 @@ async function parseResumableCandidate(args: {
|
||||
}
|
||||
return {
|
||||
value: entry,
|
||||
// Why: the index must see the state without the partial line. finalize only
|
||||
// reads the accumulator (rows are emitted in consumeLine), so this second
|
||||
// call recomputes metadata without re-emitting any captured message.
|
||||
session: displayState === state ? session : await state.finalize(args.platform),
|
||||
byteOffset: readResult.consumedThrough
|
||||
}
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
import { beforeAll, expect, it, vi } from 'vitest'
|
||||
import { AI_VAULT_SERVICE_PROTOCOL_VERSION } from './session-scanner-service-protocol'
|
||||
|
||||
// The service entry's stdio is piped to the parent's console
|
||||
// (session-scanner-service-spawn.ts), so anything it logs leaves the child.
|
||||
|
||||
const INDEX_PATH = '/Users/somebody/Library/orca/session-index.sqlite'
|
||||
|
||||
vi.mock('../ai-vault-search/session-search-service', () => ({
|
||||
SessionSearchService: class {
|
||||
constructor() {
|
||||
throw new Error(`unable to open database file ${INDEX_PATH}`)
|
||||
}
|
||||
}
|
||||
}))
|
||||
vi.mock('./session-scanner', () => ({ scanAiVaultSessions: vi.fn() }))
|
||||
vi.mock('./session-parse-cache-persistence', () => ({
|
||||
flushSessionParseCachePersist: vi.fn(() => Promise.resolve()),
|
||||
initSessionParseCachePersistence: vi.fn()
|
||||
}))
|
||||
vi.mock('./session-subagent-reader', () => ({
|
||||
listLocalAiVaultSubagentSessions: vi.fn(() => Promise.resolve({ sessions: [], issues: [] }))
|
||||
}))
|
||||
|
||||
let logged: unknown[] = []
|
||||
|
||||
beforeAll(async () => {
|
||||
process.send = (() => true) as typeof process.send
|
||||
vi.spyOn(console, 'error').mockImplementation((...args: unknown[]) => {
|
||||
logged = args
|
||||
})
|
||||
await import('./session-scanner-service-entry')
|
||||
process.emit(
|
||||
'message',
|
||||
{
|
||||
type: 'init',
|
||||
protocol: AI_VAULT_SERVICE_PROTOCOL_VERSION,
|
||||
sessionSearch: { databasePath: INDEX_PATH, enabled: true, historyDays: null }
|
||||
} as never,
|
||||
undefined as never
|
||||
)
|
||||
})
|
||||
|
||||
it('logs only the error name when the search index cannot be opened', () => {
|
||||
expect(logged[0]).toBe('[ai-vault] session search index unavailable:')
|
||||
expect(logged[1]).toBe('Error')
|
||||
|
||||
// Neither the thrown object nor the user's index path may reach the pipe.
|
||||
for (const value of logged) {
|
||||
expect(value).not.toBeInstanceOf(Error)
|
||||
expect(String(value)).not.toContain(INDEX_PATH)
|
||||
}
|
||||
})
|
||||
@@ -186,7 +186,7 @@ async function shutdown(): Promise<void> {
|
||||
}
|
||||
await Promise.allSettled([cacheLane, interactiveLane])
|
||||
await flushSessionParseCachePersist()
|
||||
sessionSearch?.dispose()
|
||||
await sessionSearch?.close()
|
||||
process.disconnect?.()
|
||||
}
|
||||
|
||||
@@ -204,7 +204,12 @@ process.on('message', (raw: AiVaultServiceParentMessage) => {
|
||||
try {
|
||||
sessionSearch = new SessionSearchService(raw.sessionSearch)
|
||||
} catch (error) {
|
||||
console.error('[ai-vault] session search index unavailable:', error)
|
||||
// Name only: this stream is piped to the parent's console, so nothing
|
||||
// from a transcript-bearing failure may ride out on it.
|
||||
console.error(
|
||||
'[ai-vault] session search index unavailable:',
|
||||
error instanceof Error ? error.name : 'IndexOpenError'
|
||||
)
|
||||
}
|
||||
}
|
||||
send({ type: 'ready', protocol: AI_VAULT_SERVICE_PROTOCOL_VERSION, pid: process.pid })
|
||||
|
||||
@@ -10,7 +10,6 @@
|
||||
// What Node and libuv need to start and resolve a home, temp dir and locale.
|
||||
// Exported for sibling plain-node forks (the WSL transcript fs process).
|
||||
export const RUNTIME_ENV_ALLOWLIST = [
|
||||
'ORCA_BACKGROUND_LAUNCH',
|
||||
'PATH',
|
||||
'HOME',
|
||||
'USERPROFILE',
|
||||
|
||||
@@ -32,12 +32,10 @@ export type SessionSearchIndexUpdate = {
|
||||
|
||||
export type SessionSearchIndexResult = Pick<SessionSearchIndexUpdate, 'session' | 'byteOffset'>
|
||||
|
||||
/** Streaming writes receive final metadata only when parsing completes. */
|
||||
export type SessionSearchIndexWrite =
|
||||
| SessionSearchIndexUpdate
|
||||
| (Omit<SessionSearchIndexUpdate, 'session' | 'byteOffset'> & {
|
||||
result: Promise<SessionSearchIndexResult>
|
||||
})
|
||||
/** Final metadata and cursor arrive only when the streamed parse completes. */
|
||||
export type SessionSearchIndexWrite = Omit<SessionSearchIndexUpdate, 'session' | 'byteOffset'> & {
|
||||
result: Promise<SessionSearchIndexResult>
|
||||
}
|
||||
|
||||
export type SessionSearchFileIdentity = { dev: number; ino: number } | null
|
||||
|
||||
@@ -48,7 +46,6 @@ export type SessionSearchIndexedFile = {
|
||||
}
|
||||
|
||||
export type SessionSearchIndexSink = {
|
||||
streamingCapture?: boolean
|
||||
acceptsCandidate?(candidate: SessionFileCandidate): boolean
|
||||
updateMetadata?(candidate: SessionFileCandidate, session: AiVaultSession): void
|
||||
/**
|
||||
@@ -117,14 +114,6 @@ export function withoutSessionSearchCapture<T>(fn: () => T): T {
|
||||
return captureStorage.run(null, fn)
|
||||
}
|
||||
|
||||
export async function withSessionSearchCapture<T>(
|
||||
fn: () => Promise<T>
|
||||
): Promise<{ value: T; messages: SessionSearchCapturedMessage[] }> {
|
||||
const scope = { messages: [] as SessionSearchCapturedMessage[] }
|
||||
const value = await captureStorage.run(scope, fn)
|
||||
return { value, messages: scope.messages }
|
||||
}
|
||||
|
||||
export async function checkpointSessionSearchCapture(): Promise<void> {
|
||||
const signal = getSessionSearchCaptureSignal()
|
||||
throwIfAiVaultScanCancelled(signal)
|
||||
|
||||
@@ -3,7 +3,6 @@ import type { AiVaultSession } from '../../shared/ai-vault-types'
|
||||
import type { SessionSearchIndexSink, SessionSearchIndexUpdate } from './session-search-capture'
|
||||
import {
|
||||
getSessionSearchCaptureSignal,
|
||||
withSessionSearchCapture,
|
||||
withStreamingSessionSearchCapture
|
||||
} from './session-search-capture'
|
||||
import { SessionSearchMessageChannel } from './session-search-message-channel'
|
||||
@@ -23,17 +22,6 @@ export async function captureIndexedSessionParse<T>(
|
||||
throwIfAiVaultScanCancelled(signal)
|
||||
return result
|
||||
}
|
||||
if (!sink.streamingCapture) {
|
||||
const captured = await withSessionSearchCapture(read)
|
||||
await sink.apply({
|
||||
...base,
|
||||
signal,
|
||||
session: captured.value.session,
|
||||
byteOffset: captured.value.byteOffset,
|
||||
messages: captured.messages
|
||||
})
|
||||
return captured.value.value
|
||||
}
|
||||
const channel = new SessionSearchMessageChannel()
|
||||
const stop = (): void => channel.stop()
|
||||
signal?.addEventListener('abort', stop, { once: true })
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
import { expect, it } from 'vitest'
|
||||
import { SessionSearchMessageChannel } from './session-search-message-channel'
|
||||
|
||||
const message = (text: string) => ({ role: 'user' as const, text, timestamp: null })
|
||||
|
||||
/** Resolves to 'stalled' when a producer is never woken, instead of hanging the run. */
|
||||
function settledOrStalled(promises: Promise<unknown>[]): Promise<string> {
|
||||
return Promise.race([
|
||||
Promise.all(promises).then(() => 'settled'),
|
||||
new Promise<string>((resolve) => setTimeout(() => resolve('stalled'), 250))
|
||||
])
|
||||
}
|
||||
|
||||
it('resumes every producer waiting on a checkpoint, not just the last one', async () => {
|
||||
const channel = new SessionSearchMessageChannel()
|
||||
channel.push(message('first'))
|
||||
const first = channel.checkpoint()
|
||||
channel.push(message('second'))
|
||||
const second = channel.checkpoint()
|
||||
const drained: string[] = []
|
||||
const consumer = (async () => {
|
||||
for await (const value of channel) {
|
||||
drained.push(value.text)
|
||||
}
|
||||
})()
|
||||
channel.close()
|
||||
await consumer
|
||||
expect(drained).toEqual(['first', 'second'])
|
||||
expect(await settledOrStalled([first, second])).toBe('settled')
|
||||
})
|
||||
|
||||
it('releases checkpoint waiters when the consumer stops', async () => {
|
||||
const channel = new SessionSearchMessageChannel()
|
||||
channel.push(message('first'))
|
||||
const first = channel.checkpoint()
|
||||
channel.push(message('second'))
|
||||
const second = channel.checkpoint()
|
||||
channel.stop()
|
||||
expect(await settledOrStalled([first, second])).toBe('settled')
|
||||
})
|
||||
@@ -4,7 +4,7 @@ import type { SessionSearchCapturedMessage } from './session-search-capture'
|
||||
export class SessionSearchMessageChannel implements AsyncIterable<SessionSearchCapturedMessage> {
|
||||
private queued: SessionSearchCapturedMessage[] = []
|
||||
private wake: (() => void) | null = null
|
||||
private drained: (() => void) | null = null
|
||||
private drained: (() => void)[] = []
|
||||
private ended = false
|
||||
private stopped = false
|
||||
private failure: unknown
|
||||
@@ -19,8 +19,10 @@ export class SessionSearchMessageChannel implements AsyncIterable<SessionSearchC
|
||||
if (!this.queued.length || this.stopped) {
|
||||
return Promise.resolve()
|
||||
}
|
||||
// Why: concurrent producers each need their own resolver; a single slot
|
||||
// would strand every waiter but the last one inside its parse.
|
||||
return new Promise((resolve) => {
|
||||
this.drained = resolve
|
||||
this.drained.push(resolve)
|
||||
})
|
||||
}
|
||||
close(error?: unknown): void {
|
||||
@@ -31,9 +33,17 @@ export class SessionSearchMessageChannel implements AsyncIterable<SessionSearchC
|
||||
stop(): void {
|
||||
this.stopped = true
|
||||
this.queued = []
|
||||
this.drained?.()
|
||||
this.releaseDrained()
|
||||
this.wake?.()
|
||||
}
|
||||
private releaseDrained(): void {
|
||||
const waiting = this.drained
|
||||
this.drained = []
|
||||
for (const resolve of waiting) {
|
||||
resolve()
|
||||
}
|
||||
}
|
||||
|
||||
async *[Symbol.asyncIterator](): AsyncGenerator<SessionSearchCapturedMessage> {
|
||||
while (!this.stopped) {
|
||||
const batch = this.queued
|
||||
@@ -41,8 +51,7 @@ export class SessionSearchMessageChannel implements AsyncIterable<SessionSearchC
|
||||
for (const message of batch) {
|
||||
yield message
|
||||
}
|
||||
this.drained?.()
|
||||
this.drained = null
|
||||
this.releaseDrained()
|
||||
if (this.failure) {
|
||||
throw this.failure
|
||||
}
|
||||
|
||||
@@ -34,7 +34,7 @@ it('cancels a queued parse without terminating the ordinary list ahead of it', a
|
||||
expect(getEventListeners(controller.signal, 'abort')).toHaveLength(0)
|
||||
worker.emit('message', {
|
||||
id: worker.requests[0].id,
|
||||
ok: true,
|
||||
kind: 'result',
|
||||
value: { candidates: [], issues: [] }
|
||||
})
|
||||
expect(await list).toEqual([])
|
||||
@@ -63,9 +63,9 @@ it('cancels during backpressure without a late ack and allows the queued list to
|
||||
const list = client.list({ dbPaths: [args.dbPath], limit: 10, issues: [] })
|
||||
workers[0].emit('message', {
|
||||
id: workers[0].requests[0].id,
|
||||
ok: true,
|
||||
captureBatch: 1,
|
||||
value: []
|
||||
kind: 'batch',
|
||||
batch: 1,
|
||||
messages: []
|
||||
})
|
||||
expect(checkpoint).toHaveBeenCalledOnce()
|
||||
controller.abort()
|
||||
@@ -78,7 +78,7 @@ it('cancels during backpressure without a late ack and allows the queued list to
|
||||
expect(getEventListeners(controller.signal, 'abort')).toHaveLength(0)
|
||||
workers[1].emit('message', {
|
||||
id: workers[1].requests[0].id,
|
||||
ok: true,
|
||||
kind: 'result',
|
||||
value: { candidates: [], issues: [] }
|
||||
})
|
||||
expect(await list).toEqual([])
|
||||
@@ -90,7 +90,7 @@ it('removes the cancellation listener after success and rejects pre-aborted work
|
||||
const client = new OpenCodeSqliteWorkerClient({ workerFactory: factory })
|
||||
const controller = new AbortController()
|
||||
const parsed = withSessionSearchIndexRequired(() => client.parse(args), controller.signal)
|
||||
worker.emit('message', { id: worker.requests[0].id, ok: true, value: { session: null } })
|
||||
worker.emit('message', { id: worker.requests[0].id, kind: 'result', value: { session: null } })
|
||||
expect(await parsed).toBeNull()
|
||||
expect(getEventListeners(controller.signal, 'abort')).toHaveLength(0)
|
||||
controller.abort()
|
||||
@@ -126,7 +126,6 @@ it('unblocks a local capture checkpoint on abort while an ordinary list still co
|
||||
previousByteOffset: 0
|
||||
}
|
||||
const sink = {
|
||||
streamingCapture: true,
|
||||
indexedFile: () => null,
|
||||
markStale() {},
|
||||
async apply() {
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
import type { Worker } from 'node:worker_threads'
|
||||
import type { OpenCodeSqliteCaptureBatch } from './session-scanner-opencode-sqlite-worker-protocol'
|
||||
import { AsyncResource } from 'node:async_hooks'
|
||||
import {
|
||||
captureSessionSearchMessage,
|
||||
checkpointSessionSearchCapture,
|
||||
type SessionSearchCapturedMessage
|
||||
} from './session-search-capture'
|
||||
|
||||
// The main-thread end of the worker's capture batch/ack loop: one batch in
|
||||
// flight, acknowledged only once the caller's sink has taken it.
|
||||
|
||||
export type OpenCodeCaptureConsumer = (messages: SessionSearchCapturedMessage[]) => Promise<void>
|
||||
|
||||
/** Binds the current capture scope: AsyncLocalStorage does not survive the worker hop. */
|
||||
export function bindOpenCodeCaptureConsumer(): OpenCodeCaptureConsumer {
|
||||
return AsyncResource.bind(async (messages: SessionSearchCapturedMessage[]) => {
|
||||
for (const message of messages) {
|
||||
captureSessionSearchMessage(message)
|
||||
}
|
||||
await checkpointSessionSearchCapture()
|
||||
})
|
||||
}
|
||||
|
||||
type DeadlinedCall = {
|
||||
capture?: OpenCodeCaptureConsumer
|
||||
timer: NodeJS.Timeout | null
|
||||
timeoutMs: number
|
||||
}
|
||||
|
||||
// Reset rather than cleared: total production time stays unbounded (that is the
|
||||
// point of the credit loop), but each individual stall is still capped, so a
|
||||
// backlogged index writer costs one scan issue instead of wedging the client.
|
||||
function restartDeadline(call: DeadlinedCall, onTimeout: () => void): void {
|
||||
if (call.timer) {
|
||||
clearTimeout(call.timer)
|
||||
}
|
||||
call.timer = setTimeout(onTimeout, call.timeoutMs)
|
||||
call.timer.unref?.()
|
||||
}
|
||||
|
||||
export function receiveOpenCodeCaptureBatch(args: {
|
||||
call: DeadlinedCall
|
||||
batch: OpenCodeSqliteCaptureBatch
|
||||
worker: Worker | null
|
||||
isActive: () => boolean
|
||||
onTimeout: () => void
|
||||
onProtocolViolation: (error: Error) => void
|
||||
onConsumerError: (error: Error) => void
|
||||
}): void {
|
||||
const { call } = args
|
||||
if (!call.capture) {
|
||||
args.onProtocolViolation(new Error('Unexpected OpenCode capture batch.'))
|
||||
return
|
||||
}
|
||||
restartDeadline(call, args.onTimeout)
|
||||
void call
|
||||
.capture(args.batch.messages)
|
||||
.then(() => {
|
||||
if (!args.isActive()) {
|
||||
return
|
||||
}
|
||||
restartDeadline(call, args.onTimeout)
|
||||
args.worker?.postMessage({
|
||||
id: args.batch.id,
|
||||
kind: 'captureAck',
|
||||
batch: args.batch.batch
|
||||
})
|
||||
})
|
||||
.catch((error) => {
|
||||
if (args.isActive()) {
|
||||
args.onConsumerError(error instanceof Error ? error : new Error(String(error)))
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -1,33 +1,76 @@
|
||||
import type SyncDatabase from '../sqlite/sync-database'
|
||||
import { captureIndexableText, toolCallText } from './session-search-content'
|
||||
import { canReadOpenCodeMessageParts } from './session-scanner-opencode-sqlite-schema'
|
||||
import {
|
||||
checkpointSessionSearchCapture,
|
||||
isSessionSearchCaptureActive
|
||||
} from './session-search-capture'
|
||||
|
||||
const SESSION_PARTS_SQL = `SELECT json_extract(m.data, '$.role') AS role,
|
||||
p.data AS data, p.time_created AS ts FROM message m JOIN part p ON p.message_id = m.id
|
||||
WHERE m.session_id = ? ORDER BY m.time_created, m.id, p.time_created, p.id`
|
||||
|
||||
type OpenCodePartRow = {
|
||||
role: unknown
|
||||
part: {
|
||||
type?: string
|
||||
text?: string
|
||||
tool?: string
|
||||
state?: { input?: unknown; output?: string }
|
||||
}
|
||||
ts: unknown
|
||||
}
|
||||
|
||||
function decodePart(data: unknown): OpenCodePartRow['part'] | null {
|
||||
try {
|
||||
const parsed = JSON.parse(String(data)) as unknown
|
||||
return parsed && typeof parsed === 'object' && !Array.isArray(parsed)
|
||||
? (parsed as OpenCodePartRow['part'])
|
||||
: null
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Yield every decodable part of one session.
|
||||
*
|
||||
* A read failure ends the stream instead of throwing: search coverage degrades
|
||||
* to no rows for this session, which must never cost the session its place in
|
||||
* the list. Consumer-thrown cancellation resumes the generator with a `return`
|
||||
* completion, so it never reaches the catch and still propagates to the caller.
|
||||
*/
|
||||
function* readOpenCodeSessionParts(
|
||||
db: SyncDatabase,
|
||||
sessionId: string
|
||||
): Generator<OpenCodePartRow> {
|
||||
try {
|
||||
for (const row of db.prepare(SESSION_PARTS_SQL).iterate(sessionId)) {
|
||||
const part = decodePart(row.data)
|
||||
if (part) {
|
||||
yield { role: row.role, part, ts: row.ts }
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
console.warn(
|
||||
'[ai-vault] opencode search capture skipped',
|
||||
error instanceof Error ? error.name : 'ReadError'
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/** The preview ring is deliberately small; search consumes every part once. */
|
||||
export async function captureOpenCodeSession(db: SyncDatabase, sessionId: string): Promise<void> {
|
||||
if (!isSessionSearchCaptureActive()) {
|
||||
if (!isSessionSearchCaptureActive() || !canReadOpenCodeMessageParts(db)) {
|
||||
return
|
||||
}
|
||||
const rows = db
|
||||
.prepare(`SELECT json_extract(m.data, '$.role') AS role,
|
||||
p.data AS data, p.time_created AS ts FROM message m JOIN part p ON p.message_id = m.id
|
||||
WHERE m.session_id = ? ORDER BY m.time_created, m.id, p.time_created, p.id`)
|
||||
.iterate(sessionId)
|
||||
for (const row of rows) {
|
||||
const part = JSON.parse(String(row.data)) as {
|
||||
type?: string
|
||||
text?: string
|
||||
tool?: string
|
||||
state?: { input?: unknown; output?: string }
|
||||
}
|
||||
for (const { role, part, ts } of readOpenCodeSessionParts(db, sessionId)) {
|
||||
if (part.type === 'text' && typeof part.text === 'string') {
|
||||
captureIndexableText(row.role === 'user' ? 'user' : 'assistant', part.text, row.ts)
|
||||
captureIndexableText(role === 'user' ? 'user' : 'assistant', part.text, ts)
|
||||
} else if (part.type === 'tool') {
|
||||
captureIndexableText('tool', toolCallText(part.tool, part.state?.input), row.ts)
|
||||
captureIndexableText('tool', toolCallText(part.tool, part.state?.input), ts)
|
||||
if (typeof part.state?.output === 'string') {
|
||||
captureIndexableText('tool', part.state.output, row.ts)
|
||||
captureIndexableText('tool', part.state.output, ts)
|
||||
}
|
||||
}
|
||||
await checkpointSessionSearchCapture()
|
||||
|
||||
@@ -39,7 +39,7 @@ export class OpenCodeWorkerSearchCapture {
|
||||
try {
|
||||
await new Promise<void>((resolve) => {
|
||||
this.acknowledgeBatch = resolve
|
||||
this.send({ id: this.id, ok: true, captureBatch: sequence, value: messages })
|
||||
this.send({ id: this.id, kind: 'batch', batch: sequence, messages })
|
||||
})
|
||||
} finally {
|
||||
this.acknowledgeBatch = null
|
||||
|
||||
@@ -1,94 +0,0 @@
|
||||
import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation'
|
||||
import type { Worker } from 'node:worker_threads'
|
||||
import type {
|
||||
OpenCodeSqliteWorkerRequest,
|
||||
OpenCodeSqliteWorkerResponse
|
||||
} from './session-scanner-opencode-sqlite-worker-protocol'
|
||||
import { AsyncResource } from 'node:async_hooks'
|
||||
import {
|
||||
captureSessionSearchMessage,
|
||||
checkpointSessionSearchCapture,
|
||||
getSessionSearchCaptureSignal,
|
||||
type SessionSearchCapturedMessage
|
||||
} from './session-search-capture'
|
||||
|
||||
export type OpenCodeCaptureConsumer = (messages: SessionSearchCapturedMessage[]) => Promise<void>
|
||||
|
||||
export function bindOpenCodeCaptureConsumer(): OpenCodeCaptureConsumer {
|
||||
return AsyncResource.bind(async (messages: SessionSearchCapturedMessage[]) => {
|
||||
for (const message of messages) {
|
||||
captureSessionSearchMessage(message)
|
||||
}
|
||||
await checkpointSessionSearchCapture()
|
||||
})
|
||||
}
|
||||
|
||||
export function receiveOpenCodeCaptureBatch(args: {
|
||||
call: { capture?: OpenCodeCaptureConsumer; timer: NodeJS.Timeout | null; timeoutMs: number }
|
||||
response: Extract<OpenCodeSqliteWorkerResponse, { ok: true }>
|
||||
worker: Worker | null
|
||||
isActive: () => boolean
|
||||
onTimeout: () => void
|
||||
onError: (error: Error) => void
|
||||
}): void {
|
||||
const { call } = args
|
||||
if (!call.capture) {
|
||||
args.onError(new Error('Unexpected OpenCode capture batch.'))
|
||||
return
|
||||
}
|
||||
// Backpressure belongs to the writer; the worker deadline covers time spent producing.
|
||||
if (call.timer) {
|
||||
clearTimeout(call.timer)
|
||||
call.timer = null
|
||||
}
|
||||
void call
|
||||
.capture(args.response.value as SessionSearchCapturedMessage[])
|
||||
.then(() => {
|
||||
if (!args.isActive()) {
|
||||
return
|
||||
}
|
||||
call.timer = setTimeout(args.onTimeout, call.timeoutMs)
|
||||
call.timer.unref?.()
|
||||
args.worker?.postMessage({
|
||||
id: args.response.id,
|
||||
kind: 'captureAck',
|
||||
batch: args.response.captureBatch
|
||||
})
|
||||
})
|
||||
.catch((error) => {
|
||||
if (args.isActive()) {
|
||||
args.onError(error instanceof Error ? error : new Error(String(error)))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/** The request owns its abort listener until either queued or active work settles. */
|
||||
export function bindOpenCodeCaptureCancellation(
|
||||
resolve: (value: unknown) => void,
|
||||
reject: (error: Error) => void,
|
||||
cancel: () => void
|
||||
): { resolve: typeof resolve; reject: typeof reject } {
|
||||
const signal = getSessionSearchCaptureSignal()
|
||||
throwIfAiVaultScanCancelled(signal)
|
||||
signal?.addEventListener('abort', cancel, { once: true })
|
||||
const cleanup = (): void => signal?.removeEventListener('abort', cancel)
|
||||
return {
|
||||
resolve: (value) => {
|
||||
cleanup()
|
||||
resolve(value)
|
||||
},
|
||||
reject: (error) => {
|
||||
cleanup()
|
||||
reject(error)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export type OpenCodePendingCall = {
|
||||
request: OpenCodeSqliteWorkerRequest
|
||||
timeoutMs: number
|
||||
resolve: (value: unknown) => void
|
||||
reject: (error: Error) => void
|
||||
timer: NodeJS.Timeout | null
|
||||
capture?: OpenCodeCaptureConsumer
|
||||
}
|
||||
@@ -19,6 +19,7 @@ import {
|
||||
mergeAiVaultListResults
|
||||
} from '../ai-vault/session-list-results'
|
||||
import type { AiVaultSearchArgs } from '../../shared/ai-vault-search-types'
|
||||
import { projectSessionSearchResult } from '../../shared/ai-vault-search-projection'
|
||||
import { scanSshAiVaultSessions } from '../ai-vault/ssh-session-list'
|
||||
import { AiVaultScanCoordinator } from '../ai-vault/ai-vault-scan-coordinator'
|
||||
import type { AiVaultDeleteSessionArgs } from '../../shared/ai-vault-session-deletion'
|
||||
@@ -260,9 +261,10 @@ export function registerAiVaultHandlers(options: AiVaultHandlerOptions = {}): vo
|
||||
}
|
||||
})
|
||||
// Local-only: the search index is built beside the transcripts on this host,
|
||||
// so a remote scope has nothing to consult here.
|
||||
// so a remote scope has nothing to consult here. Projected all the same, so the
|
||||
// renderer sees one result shape whether the host is local or remote.
|
||||
ipcMain.handle('aiVault:searchSessions', (_event, args: AiVaultSearchArgs) =>
|
||||
searchAiVaultSessions(args)
|
||||
searchAiVaultSessions(args).then(projectSessionSearchResult)
|
||||
)
|
||||
ipcMain.handle('aiVault:searchCoverage', () => readAiVaultSearchCoverage())
|
||||
ipcMain.handle('aiVault:searchIndexSize', () => ({
|
||||
|
||||
@@ -37,7 +37,7 @@ import {
|
||||
normalizeComputerAwakeMode
|
||||
} from '../../shared/computer-awake-mode'
|
||||
import {
|
||||
applyAiVaultSearchSettings,
|
||||
applyAiVaultSearchSettingsChange,
|
||||
installAiVaultSearchSettingsSource
|
||||
} from '../ai-vault-search/session-search-enablement'
|
||||
import { resolveAiVaultSearchSettings } from '../../shared/ai-vault-search-settings'
|
||||
@@ -278,9 +278,9 @@ export function registerSettingsHandlers(
|
||||
applyAppIcon(result.appIcon)
|
||||
}
|
||||
if ('aiVaultSearch' in sanitizedArgs) {
|
||||
await applyAiVaultSearchSettings(result, {
|
||||
persist: () => store.flushPendingOrThrowAsync({ drainToStableGeneration: false })
|
||||
})
|
||||
applyAiVaultSearchSettingsChange(before, result, () =>
|
||||
store.flushPendingOrThrowAsync({ drainToStableGeneration: false })
|
||||
)
|
||||
}
|
||||
|
||||
// Why: telemetry-plan.md§Settings — fire `settings_changed` only for
|
||||
|
||||
@@ -269,14 +269,44 @@ describe('aiVault.searchCoverage handler', () => {
|
||||
).resolves.toMatchObject({ ok: true, result: COVERAGE })
|
||||
expect(readAiVaultSearchCoverage).toHaveBeenCalledWith(controller.signal)
|
||||
})
|
||||
})
|
||||
|
||||
it('rejects a non-runtime execution host id', async () => {
|
||||
describe('host-local execution boundary', () => {
|
||||
// Every one of these runs on this host's own index. An id naming a host this
|
||||
// process does not execute on must be refused, not answered locally.
|
||||
const methods = [
|
||||
['aiVault.searchSessions', { query: 'q' }],
|
||||
['aiVault.searchCoverage', {}],
|
||||
['aiVault.searchIndexStatus', {}],
|
||||
['aiVault.configureSessionSearch', { enabled: true }]
|
||||
] as const
|
||||
|
||||
it.each(['ssh:build-server', 'local', 'not-a-host'])(
|
||||
'refuses %s on every search method instead of answering with this host',
|
||||
async (executionHostId) => {
|
||||
const dispatcher = makeDispatcher()
|
||||
for (const [method, params] of methods) {
|
||||
await expect(
|
||||
dispatcher.dispatch(makeRequest(method, { ...params, executionHostId }))
|
||||
).resolves.toMatchObject({ ok: false })
|
||||
}
|
||||
expect(searchAiVaultSessions).not.toHaveBeenCalled()
|
||||
expect(readAiVaultSearchCoverage).not.toHaveBeenCalled()
|
||||
expect(readAiVaultSearchIndexStatus).not.toHaveBeenCalled()
|
||||
expect(configureAiVaultSessionSearch).not.toHaveBeenCalled()
|
||||
}
|
||||
)
|
||||
|
||||
it('accepts a runtime id on every search method without letting it route the call', async () => {
|
||||
const dispatcher = makeDispatcher()
|
||||
|
||||
await expect(
|
||||
dispatcher.dispatch(makeRequest('aiVault.searchCoverage', { executionHostId: 'ssh:box' }))
|
||||
).resolves.toMatchObject({ ok: false })
|
||||
expect(readAiVaultSearchCoverage).not.toHaveBeenCalled()
|
||||
for (const [method, params] of methods) {
|
||||
await expect(
|
||||
dispatcher.dispatch(makeRequest(method, { ...params, executionHostId: 'runtime:env-1' }))
|
||||
).resolves.toMatchObject({ ok: true })
|
||||
}
|
||||
expect(searchAiVaultSessions.mock.calls[0]?.[0]).not.toHaveProperty('executionHostId')
|
||||
expect(configureAiVaultSessionSearch.mock.calls[0]?.[0]).not.toHaveProperty('executionHostId')
|
||||
expect(sshSearchAiVault).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ import { AI_VAULT_SESSION_TITLE_REQUEST_MAX_COUNT } from '../../../../shared/ai-
|
||||
import type { AiVaultPrepareSessionResumeArgs } from '../../../../shared/ai-vault-resume-preparation'
|
||||
import { LOCAL_EXECUTION_HOST_ID, parseExecutionHostId } from '../../../../shared/execution-host'
|
||||
import { describeAiVaultScanError } from '../../../../shared/ai-vault-scan-error-message'
|
||||
import { SESSION_SEARCH_METHODS } from '../../../../shared/ai-vault-search-rpc-methods'
|
||||
import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version'
|
||||
import {
|
||||
assertLegacyAiVaultResumeAllowed,
|
||||
@@ -23,6 +24,13 @@ import {
|
||||
const AI_VAULT_SCOPE_PATH_MAX_LENGTH = 4096
|
||||
const AI_VAULT_LIMIT_MAX = 2000
|
||||
|
||||
// Why: this is the whole SSH/foreign-host boundary for the aiVault surface. The
|
||||
// scan and the index are host-local, so a caller must not be able to name a host
|
||||
// this process does not execute on — an `ssh:` or `local` id is refused here
|
||||
// rather than silently answered with this runtime's own transcripts
|
||||
// (docs/reference/ssh-execution-boundary.md rule 1). A `runtime:` id names the
|
||||
// *client's* saved environment, whose id this host never learns, so it is
|
||||
// accepted for restamping only and never routes anything.
|
||||
const executionHostIdSchema = z.string().transform((value, ctx): `runtime:${string}` => {
|
||||
const parsed = parseExecutionHostId(value)
|
||||
if (parsed?.kind === 'runtime') {
|
||||
@@ -102,25 +110,25 @@ export const AiVaultConfigureSessionSearchParams = SessionSearchConfigureSchema.
|
||||
|
||||
export const AI_VAULT_METHODS: RpcMethod[] = [
|
||||
defineMethod({
|
||||
name: 'aiVault.sshSearchSessions',
|
||||
name: SESSION_SEARCH_METHODS.query.runtimeSsh,
|
||||
params: SessionSearchQuerySchema.extend({ targetId: z.string().min(1).max(512) }),
|
||||
handler: ({ targetId, ...params }, { runtime, signal }) =>
|
||||
runtime.sshSearchAiVault(targetId, 'query', params, signal)
|
||||
}),
|
||||
defineMethod({
|
||||
name: 'aiVault.sshSearchIndexStatus',
|
||||
name: SESSION_SEARCH_METHODS.status.runtimeSsh,
|
||||
params: z.object({ targetId: z.string().min(1).max(512) }),
|
||||
handler: ({ targetId }, { runtime, signal }) =>
|
||||
runtime.sshSearchAiVault(targetId, 'status', {}, signal)
|
||||
}),
|
||||
defineMethod({
|
||||
name: 'aiVault.sshSearchConfigure',
|
||||
name: SESSION_SEARCH_METHODS.configure.runtimeSsh,
|
||||
params: SessionSearchConfigureSchema.extend({ targetId: z.string().min(1).max(512) }),
|
||||
handler: ({ targetId, ...params }, { runtime, signal }) =>
|
||||
runtime.sshSearchAiVault(targetId, 'configure', params, signal)
|
||||
}),
|
||||
defineMethod({
|
||||
name: 'aiVault.searchSessions',
|
||||
name: SESSION_SEARCH_METHODS.query.runtime,
|
||||
params: AiVaultSearchSessionsParams,
|
||||
// Why: the index lives with the transcripts, so this runs on the host the
|
||||
// client addressed; the id only names that host, it never redirects the search.
|
||||
@@ -133,12 +141,12 @@ export const AI_VAULT_METHODS: RpcMethod[] = [
|
||||
handler: (_params, { runtime, signal }) => runtime.readAiVaultSearchCoverage(signal)
|
||||
}),
|
||||
defineMethod({
|
||||
name: 'aiVault.searchIndexStatus',
|
||||
name: SESSION_SEARCH_METHODS.status.runtime,
|
||||
params: z.object({ executionHostId: executionHostIdSchema.optional() }),
|
||||
handler: (_params, { runtime }) => runtime.readAiVaultSearchIndexStatus()
|
||||
}),
|
||||
defineMethod({
|
||||
name: 'aiVault.configureSessionSearch',
|
||||
name: SESSION_SEARCH_METHODS.configure.runtime,
|
||||
params: AiVaultConfigureSessionSearchParams,
|
||||
// Why: consent is per machine and the index lives with the transcripts, so
|
||||
// this writes the addressed host's own setting; the id never redirects it.
|
||||
|
||||
@@ -35,9 +35,17 @@ import {
|
||||
SessionSearchQuerySchema,
|
||||
type SessionSearchConfigure
|
||||
} from '../../shared/ai-vault-search-contract'
|
||||
import {
|
||||
SESSION_SEARCH_METHODS,
|
||||
type SessionSearchOperation
|
||||
} from '../../shared/ai-vault-search-rpc-methods'
|
||||
|
||||
export type AiVaultSessionSearchConfigureArgs = SessionSearchConfigure
|
||||
|
||||
// Why: `reason` is optional on the wire type, so a host that omits it must not
|
||||
// surface an `undefined` message.
|
||||
const SEARCH_UNAVAILABLE_MESSAGE = 'Session search is unavailable on this host.'
|
||||
|
||||
export class RuntimeAiVaultCommands {
|
||||
constructor(
|
||||
private readonly getPrepareResume: () =>
|
||||
@@ -54,15 +62,18 @@ export class RuntimeAiVaultCommands {
|
||||
|
||||
search(args: AiVaultSearchArgs, signal?: AbortSignal): Promise<AiVaultSearchResult> {
|
||||
const status = this.searchIndexStatus()
|
||||
if (status.available === false || status.applied === false) {
|
||||
throw new Error(status.reason)
|
||||
// Why: an in-flight policy apply leaves `applied` false while the existing
|
||||
// index is still valid; refusing there would fail every query issued during
|
||||
// a settings write. `searchIndexStatus` remains the channel for that.
|
||||
if (status.available === false) {
|
||||
throw new Error(status.reason ?? SEARCH_UNAVAILABLE_MESSAGE)
|
||||
}
|
||||
return searchAiVaultSessions(args, { signal }).then(projectSessionSearchResult)
|
||||
}
|
||||
|
||||
async sshSearch(
|
||||
targetId: string,
|
||||
operation: 'query' | 'status' | 'configure',
|
||||
operation: SessionSearchOperation,
|
||||
args: unknown,
|
||||
signal?: AbortSignal
|
||||
): Promise<unknown> {
|
||||
@@ -77,11 +88,10 @@ export class RuntimeAiVaultCommands {
|
||||
? SessionSearchConfigureSchema.parse(args)
|
||||
: {}
|
||||
try {
|
||||
return await provider.requestHostRpc(
|
||||
`aiVault.search${{ query: 'Sessions', status: 'IndexStatus', configure: 'Configure' }[operation]}`,
|
||||
params,
|
||||
{ signal, timeoutMs: 15_000 }
|
||||
)
|
||||
return await provider.requestHostRpc(SESSION_SEARCH_METHODS[operation].relay, params, {
|
||||
signal,
|
||||
timeoutMs: 15_000
|
||||
})
|
||||
} catch (error) {
|
||||
if (
|
||||
operation === 'status' &&
|
||||
@@ -129,7 +139,7 @@ export class RuntimeAiVaultCommands {
|
||||
}
|
||||
const status = this.searchIndexStatus()
|
||||
if (status.available === false) {
|
||||
throw new Error(status.reason)
|
||||
throw new Error(status.reason ?? SEARCH_UNAVAILABLE_MESSAGE)
|
||||
}
|
||||
const current = resolveAiVaultSearchSettings(store.getSettings())
|
||||
const next = {
|
||||
|
||||
@@ -8,15 +8,24 @@ const apply = vi.hoisted(() =>
|
||||
return null
|
||||
})
|
||||
)
|
||||
vi.mock('../ai-vault-search/session-search-enablement', () => ({
|
||||
applyAiVaultSearchSettings: apply,
|
||||
readAiVaultSearchIndexStatus: () => ({
|
||||
const indexStatus = vi.hoisted(() => ({
|
||||
value: {
|
||||
enabled: true,
|
||||
historyDays: null,
|
||||
indexSizeBytes: 0,
|
||||
available: true,
|
||||
applied: true
|
||||
})
|
||||
} as Record<string, unknown>
|
||||
}))
|
||||
const searchAiVaultSessions = vi.hoisted(() => vi.fn())
|
||||
vi.mock('../ai-vault-search/session-search-enablement', () => ({
|
||||
applyAiVaultSearchSettings: apply,
|
||||
readAiVaultSearchIndexStatus: () => indexStatus.value
|
||||
}))
|
||||
vi.mock('../ai-vault/cached-session-list', () => ({
|
||||
listAiVaultSessions: vi.fn(),
|
||||
readAiVaultSearchCoverage: vi.fn(),
|
||||
searchAiVaultSessions
|
||||
}))
|
||||
|
||||
it('does not acknowledge enabling until the durable store barrier completes', async () => {
|
||||
@@ -59,3 +68,43 @@ it('reports persistence failure instead of returning a successful policy acknowl
|
||||
)
|
||||
await expect(commands.configureSearch({ enabled: true })).rejects.toThrow('disk full')
|
||||
})
|
||||
|
||||
it('answers from the existing index while a policy apply is still in flight', async () => {
|
||||
const coverage = {
|
||||
enabled: true,
|
||||
sessionsIndexed: 1,
|
||||
messagesIndexed: 1,
|
||||
providers: [],
|
||||
backfill: 'complete' as const,
|
||||
filesPending: 0,
|
||||
lastIndexedAt: null
|
||||
}
|
||||
searchAiVaultSessions.mockResolvedValue({ hits: [], route: 'and', durationMs: 1, coverage })
|
||||
const commands = new RuntimeAiVaultCommands(() => null)
|
||||
indexStatus.value = {
|
||||
enabled: true,
|
||||
historyDays: null,
|
||||
indexSizeBytes: 0,
|
||||
available: true,
|
||||
applied: false,
|
||||
reason: 'Index policy application or persistence failed or is pending.'
|
||||
}
|
||||
|
||||
await expect(commands.search({ query: 'needle' })).resolves.toMatchObject({ coverage })
|
||||
expect(searchAiVaultSessions).toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('refuses a query when the index is unavailable, with a message even if the host omits one', async () => {
|
||||
searchAiVaultSessions.mockClear()
|
||||
const commands = new RuntimeAiVaultCommands(() => null)
|
||||
indexStatus.value = {
|
||||
enabled: false,
|
||||
historyDays: null,
|
||||
indexSizeBytes: null,
|
||||
available: false,
|
||||
applied: false
|
||||
}
|
||||
|
||||
expect(() => commands.search({ query: 'needle' })).toThrow(/unavailable on this host/)
|
||||
expect(searchAiVaultSessions).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
@@ -317,7 +317,8 @@ describe('AiVaultHandler', () => {
|
||||
hostPlatform: getRemoteHostPlatform('linux-x64'),
|
||||
service: {
|
||||
listSessions: () => Promise.reject(new Error('sidecar crashed')),
|
||||
resolveSessionTitles: () => Promise.resolve({ titles: [] })
|
||||
resolveSessionTitles: () => Promise.resolve({ titles: [] }),
|
||||
search: unwiredSearch
|
||||
}
|
||||
})
|
||||
|
||||
@@ -338,7 +339,8 @@ describe('AiVaultHandler', () => {
|
||||
hostPlatform: getRemoteHostPlatform('linux-x64'),
|
||||
service: {
|
||||
listSessions: () => Promise.resolve(emptyResult()),
|
||||
resolveSessionTitles: () => Promise.reject(new Error('sidecar crashed'))
|
||||
resolveSessionTitles: () => Promise.reject(new Error('sidecar crashed')),
|
||||
search: unwiredSearch
|
||||
}
|
||||
})
|
||||
|
||||
@@ -365,7 +367,8 @@ describe('AiVaultHandler', () => {
|
||||
const error = new Error('The operation was aborted.')
|
||||
error.name = 'AbortError'
|
||||
return Promise.reject(error)
|
||||
}
|
||||
},
|
||||
search: unwiredSearch
|
||||
}
|
||||
})
|
||||
|
||||
@@ -403,10 +406,14 @@ function createTestService(
|
||||
signal
|
||||
}),
|
||||
resolveSessionTitles: (requests, signal) =>
|
||||
readAiVaultSessionTitlesFromFiles(requests, { signal })
|
||||
readAiVaultSessionTitlesFromFiles(requests, { signal }),
|
||||
search: unwiredSearch
|
||||
}
|
||||
}
|
||||
|
||||
const unwiredSearch = (): Promise<never> =>
|
||||
Promise.reject(new Error('Search is not wired in this fixture.'))
|
||||
|
||||
function createMockDispatcher(): {
|
||||
value: RelayDispatcher
|
||||
call: (method: string, params: Record<string, unknown>, signal?: AbortSignal) => Promise<unknown>
|
||||
|
||||
@@ -18,6 +18,10 @@ import { parseUnameToRelayPlatform } from '../main/ssh/relay-protocol'
|
||||
import { relayLogLine } from './relay-diagnostic-log'
|
||||
import type { RelayDispatcher } from './dispatcher'
|
||||
import { AiVaultScanCoordinator } from '../main/ai-vault/ai-vault-scan-coordinator'
|
||||
import {
|
||||
SESSION_SEARCH_METHODS,
|
||||
SESSION_SEARCH_OPERATIONS
|
||||
} from '../shared/ai-vault-search-rpc-methods'
|
||||
import type { RelayAiVaultServiceApi } from './ai-vault-service-client-state'
|
||||
|
||||
type AiVaultHandlerOptions = {
|
||||
@@ -55,16 +59,10 @@ export class AiVaultHandler {
|
||||
dispatcher.onRequest(SSH_AI_VAULT_RESOLVE_SESSION_TITLES_METHOD, (params, context) =>
|
||||
this.resolveSessionTitles(service, params, context.signal)
|
||||
)
|
||||
if (service.search) {
|
||||
for (const [suffix, action] of [
|
||||
['Sessions', 'query'],
|
||||
['IndexStatus', 'status'],
|
||||
['Configure', 'configure']
|
||||
] as const) {
|
||||
dispatcher.onRequest(`aiVault.search${suffix}`, (params, context) =>
|
||||
service.search!(action, params, context.signal)
|
||||
)
|
||||
}
|
||||
for (const operation of SESSION_SEARCH_OPERATIONS) {
|
||||
dispatcher.onRequest(SESSION_SEARCH_METHODS[operation].relay, (params, context) =>
|
||||
service.search(operation, params, context.signal)
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ import type {
|
||||
AiVaultSessionTitleRequest,
|
||||
AiVaultSessionTitlesResult
|
||||
} from '../shared/ai-vault-session-title'
|
||||
import type { SessionSearchOperation } from '../shared/ai-vault-search-rpc-methods'
|
||||
import type { SshAiVaultRelayListParams } from '../shared/ssh-ai-vault-relay'
|
||||
import type { RemoteHostPlatform } from '../main/ssh/ssh-remote-platform'
|
||||
import {
|
||||
@@ -44,7 +45,7 @@ export function createRelayAiVaultServiceCall(args: {
|
||||
}): RelayAiVaultServiceCall {
|
||||
return {
|
||||
request: args.request,
|
||||
lane: relayAiVaultServiceLane(args.request.operation),
|
||||
lane: relayAiVaultServiceLane(args.request),
|
||||
signal: args.signal,
|
||||
forceStart: args.request.operation === 'list' && args.request.params.force === true,
|
||||
resolve: args.resolve,
|
||||
@@ -110,11 +111,7 @@ export type RelayAiVaultServiceCall = {
|
||||
}
|
||||
|
||||
export type RelayAiVaultServiceApi = {
|
||||
search?(
|
||||
action: 'query' | 'status' | 'configure',
|
||||
params: unknown,
|
||||
signal?: AbortSignal
|
||||
): Promise<unknown>
|
||||
search(action: SessionSearchOperation, params: unknown, signal?: AbortSignal): Promise<unknown>
|
||||
listSessions(params: SshAiVaultRelayListParams, signal?: AbortSignal): Promise<AiVaultListResult>
|
||||
resolveSessionTitles(
|
||||
requests: AiVaultSessionTitleRequest[],
|
||||
|
||||
@@ -84,7 +84,7 @@ export class RelayAiVaultServiceClient implements RelayAiVaultServiceApi {
|
||||
}
|
||||
}
|
||||
|
||||
search: NonNullable<RelayAiVaultServiceApi['search']> = (action, params, signal) =>
|
||||
search: RelayAiVaultServiceApi['search'] = (action, params, signal) =>
|
||||
this.request(
|
||||
{ type: 'request', id: this.nextId++, operation: 'search', action, params },
|
||||
signal
|
||||
@@ -121,7 +121,7 @@ export class RelayAiVaultServiceClient implements RelayAiVaultServiceApi {
|
||||
if (this.disposed || this.restartPolicy.restartScheduled) {
|
||||
return
|
||||
}
|
||||
for (const lane of ['cache', 'interactive'] as const) {
|
||||
for (const lane of ['cache', 'interactive', 'search'] as const) {
|
||||
if (this.active.has(lane)) {
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
isRelayAiVaultServiceRequest,
|
||||
relayAiVaultServiceLane,
|
||||
type RelayAiVaultServiceChildMessage,
|
||||
type RelayAiVaultServiceLane,
|
||||
type RelayAiVaultServiceInit,
|
||||
type RelayAiVaultServiceParentMessage,
|
||||
type RelayAiVaultServiceRequest
|
||||
@@ -22,8 +23,11 @@ const cancelled = new Set<number>()
|
||||
const pending = new Set<number>()
|
||||
const provider = createRelayAiVaultFilesystemProvider()
|
||||
let init: RelayAiVaultServiceInit | null = null
|
||||
let cacheLane = Promise.resolve()
|
||||
let interactiveLane = Promise.resolve()
|
||||
const lanes: Record<RelayAiVaultServiceLane, Promise<void>> = {
|
||||
cache: Promise.resolve(),
|
||||
interactive: Promise.resolve(),
|
||||
search: Promise.resolve()
|
||||
}
|
||||
let shuttingDown = false
|
||||
let searchOwner: RelaySessionSearchOwner | null = null
|
||||
|
||||
@@ -86,7 +90,7 @@ async function shutdown(): Promise<void> {
|
||||
for (const controller of controllers.values()) {
|
||||
controller.abort()
|
||||
}
|
||||
await Promise.allSettled([cacheLane, interactiveLane])
|
||||
await Promise.allSettled(Object.values(lanes))
|
||||
await searchOwner?.close()
|
||||
process.disconnect?.()
|
||||
}
|
||||
@@ -121,11 +125,8 @@ process.on('message', (raw: RelayAiVaultServiceParentMessage) => {
|
||||
return
|
||||
}
|
||||
pending.add(raw.id)
|
||||
if (relayAiVaultServiceLane(raw.operation) === 'interactive') {
|
||||
interactiveLane = interactiveLane.then(() => execute(raw))
|
||||
return
|
||||
}
|
||||
cacheLane = cacheLane.then(() => execute(raw))
|
||||
const lane = relayAiVaultServiceLane(raw)
|
||||
lanes[lane] = lanes[lane].then(() => execute(raw))
|
||||
})
|
||||
|
||||
process.on('disconnect', () => void shutdown())
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
import { expect, it } from 'vitest'
|
||||
import {
|
||||
isRelayAiVaultServiceRequest,
|
||||
relayAiVaultServiceLane,
|
||||
type RelayAiVaultServiceRequest
|
||||
} from './ai-vault-service-protocol'
|
||||
|
||||
const search = (action: 'query' | 'status' | 'configure'): RelayAiVaultServiceRequest => ({
|
||||
type: 'request',
|
||||
id: 1,
|
||||
operation: 'search',
|
||||
action,
|
||||
params: {}
|
||||
})
|
||||
|
||||
it('keeps a history scan and a search query off the lane that backs interactive reads', () => {
|
||||
expect(relayAiVaultServiceLane({ type: 'request', id: 1, operation: 'list', params: {} })).toBe(
|
||||
'cache'
|
||||
)
|
||||
expect(relayAiVaultServiceLane(search('query'))).toBe('search')
|
||||
expect(
|
||||
relayAiVaultServiceLane({ type: 'request', id: 1, operation: 'titles', requests: [] })
|
||||
).toBe('interactive')
|
||||
expect(relayAiVaultServiceLane(search('status'))).toBe('interactive')
|
||||
expect(relayAiVaultServiceLane(search('configure'))).toBe('interactive')
|
||||
expect(new Set([relayAiVaultServiceLane(search('query')), 'interactive']).size).toBe(2)
|
||||
})
|
||||
|
||||
it('refuses a search request whose action is not one this build owns', () => {
|
||||
expect(isRelayAiVaultServiceRequest(search('query'))).toBe(true)
|
||||
expect(
|
||||
isRelayAiVaultServiceRequest({ type: 'request', id: 1, operation: 'search', params: {} })
|
||||
).toBe(false)
|
||||
expect(
|
||||
isRelayAiVaultServiceRequest({
|
||||
type: 'request',
|
||||
id: 1,
|
||||
operation: 'search',
|
||||
action: 'drop',
|
||||
params: {}
|
||||
})
|
||||
).toBe(false)
|
||||
})
|
||||
@@ -5,6 +5,10 @@ import type {
|
||||
} from '../shared/ai-vault-session-title'
|
||||
import type { SshAiVaultRelayListParams } from '../shared/ssh-ai-vault-relay'
|
||||
import type { RemoteHostPlatform } from '../main/ssh/ssh-remote-platform'
|
||||
import {
|
||||
SESSION_SEARCH_OPERATIONS,
|
||||
type SessionSearchOperation
|
||||
} from '../shared/ai-vault-search-rpc-methods'
|
||||
|
||||
export const RELAY_AI_VAULT_SERVICE_PROTOCOL = 1
|
||||
|
||||
@@ -20,7 +24,7 @@ export type RelayAiVaultServiceRequest =
|
||||
type: 'request'
|
||||
id: number
|
||||
operation: 'search'
|
||||
action: 'query' | 'status' | 'configure'
|
||||
action: SessionSearchOperation
|
||||
params: unknown
|
||||
}
|
||||
| {
|
||||
@@ -36,14 +40,21 @@ export type RelayAiVaultServiceRequest =
|
||||
requests: AiVaultSessionTitleRequest[]
|
||||
}
|
||||
|
||||
export type RelayAiVaultServiceLane = 'cache' | 'interactive'
|
||||
export type RelayAiVaultServiceOperation = RelayAiVaultServiceRequest['operation']
|
||||
export type RelayAiVaultServiceLane = 'cache' | 'interactive' | 'search'
|
||||
|
||||
/** Queries, controls and title reads must not queue behind a full history scan. */
|
||||
/**
|
||||
* `list` is a full history scan and a search `query` can drive a backfill pass,
|
||||
* so neither may queue ahead of the interactive lane that title reads and the
|
||||
* search controls run on. Search stays correct across the split because
|
||||
* `RelaySessionSearchOwner` serializes every operation it owns.
|
||||
*/
|
||||
export function relayAiVaultServiceLane(
|
||||
operation: RelayAiVaultServiceOperation
|
||||
request: RelayAiVaultServiceRequest
|
||||
): RelayAiVaultServiceLane {
|
||||
return operation === 'list' ? 'cache' : 'interactive'
|
||||
if (request.operation === 'list') {
|
||||
return 'cache'
|
||||
}
|
||||
return request.operation === 'search' && request.action === 'query' ? 'search' : 'interactive'
|
||||
}
|
||||
|
||||
export type RelayAiVaultServiceParentMessage =
|
||||
@@ -78,7 +89,8 @@ export function isRelayAiVaultServiceRequest(value: unknown): value is RelayAiVa
|
||||
Number.isSafeInteger(message.id) &&
|
||||
(message.operation === 'list' ||
|
||||
message.operation === 'titles' ||
|
||||
message.operation === 'search')
|
||||
(message.operation === 'search' &&
|
||||
SESSION_SEARCH_OPERATIONS.includes(message.action as SessionSearchOperation)))
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
import { existsSync, lstatSync, readFileSync, statSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { writeDurableSecureJsonFile } from '../shared/secure-file'
|
||||
import {
|
||||
DEFAULT_AI_VAULT_SEARCH_SETTINGS,
|
||||
type AiVaultSearchSettings
|
||||
} from '../shared/ai-vault-search-settings'
|
||||
import { SessionSearchConfigureSchema } from '../shared/ai-vault-search-contract'
|
||||
|
||||
const POLICY_FILE_VERSION = 1
|
||||
const POLICY_FILE_MAX_BYTES = 8192
|
||||
|
||||
export const SESSION_SEARCH_POLICY_RECOVERY_HINT =
|
||||
'Run `orca search --clear-index` against this host to reset the index and policy.'
|
||||
|
||||
/** Refuses a path another account could have substituted for the real one. */
|
||||
export function assertOwnedSearchPath(path: string, directory: boolean): void {
|
||||
const stat = lstatSync(path)
|
||||
if (
|
||||
stat.isSymbolicLink() ||
|
||||
(directory ? !stat.isDirectory() : !stat.isFile()) ||
|
||||
(process.getuid && stat.uid !== process.getuid())
|
||||
) {
|
||||
throw new Error('Unsafe search owner path.')
|
||||
}
|
||||
}
|
||||
|
||||
function policyPath(directory: string): string {
|
||||
return join(directory, 'policy.json')
|
||||
}
|
||||
|
||||
/** Throws when the recorded policy cannot be vouched for; absent reads as consent-off. */
|
||||
export function readSessionSearchOwnerPolicy(
|
||||
directory: string,
|
||||
home: string
|
||||
): AiVaultSearchSettings {
|
||||
const file = policyPath(directory)
|
||||
if (!existsSync(file)) {
|
||||
return { ...DEFAULT_AI_VAULT_SEARCH_SETTINGS }
|
||||
}
|
||||
assertOwnedSearchPath(file, false)
|
||||
if (statSync(file).size > POLICY_FILE_MAX_BYTES) {
|
||||
throw new Error('Search policy exceeds its size limit.')
|
||||
}
|
||||
const saved = JSON.parse(readFileSync(file, 'utf8'))
|
||||
if (saved.home !== home || saved.version !== POLICY_FILE_VERSION) {
|
||||
throw new Error('Search source configuration changed; host policy must be reviewed.')
|
||||
}
|
||||
const policy = SessionSearchConfigureSchema.parse(saved.policy)
|
||||
if (typeof policy.enabled !== 'boolean' || policy.historyDays === undefined) {
|
||||
throw new Error('Invalid search policy.')
|
||||
}
|
||||
return {
|
||||
enabled: policy.enabled,
|
||||
historyDays: policy.historyDays,
|
||||
...(policy.paused ? { paused: true } : {})
|
||||
}
|
||||
}
|
||||
|
||||
export function writeSessionSearchOwnerPolicy(
|
||||
directory: string,
|
||||
home: string,
|
||||
policy: AiVaultSearchSettings
|
||||
): void {
|
||||
if (
|
||||
!writeDurableSecureJsonFile(policyPath(directory), {
|
||||
home,
|
||||
version: POLICY_FILE_VERSION,
|
||||
policy
|
||||
})
|
||||
) {
|
||||
throw new Error('Could not secure the host search policy.')
|
||||
}
|
||||
}
|
||||
@@ -78,11 +78,11 @@ it('finishes a slow parse before handing ownership to another relay', async () =
|
||||
let complete: (() => void) | undefined
|
||||
let parseSignal: AbortSignal | undefined
|
||||
vi.spyOn(candidateParser, 'parseSearchCandidates').mockImplementation(
|
||||
(_store, _candidates, signal) =>
|
||||
(_store, _candidates, options) =>
|
||||
new Promise<void>((resolve) => {
|
||||
complete = resolve
|
||||
parseSignal = signal
|
||||
signal?.addEventListener('abort', () => resolve(), { once: true })
|
||||
parseSignal = options?.signal
|
||||
options?.signal?.addEventListener('abort', () => resolve(), { once: true })
|
||||
})
|
||||
)
|
||||
const state = vi.spyOn(SessionSearchStore.prototype, 'setBackfillState')
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user