mirror of
https://github.com/stablyai/orca.git
synced 2026-10-08 16:02:37 +00:00
fix(ai-vault-search): address review findings and the React Doctor gate
React Doctor failed CI on a ref assigned during render in the search
request hook; the debounce now passes the serialized args into issue()
and publishes its cancel through a ref so Enter can drop both timers.
Review fixes, each with a regression test that fails when reverted:
- snippet markers are [[term]] so literal brackets in code never read as
a match, and the snippet is built from the plan that retrieved the row
(a typo-repaired hit was getting an empty snippet)
- scope paths are separator-normalized and LIKE-escaped, so a folder with
% or _ cannot widen the match; non-positive limits are clamped at the
query since the desktop IPC forwards its payload unvalidated
- a failed backfill is no longer memoized as done; widening the history
bound re-enumerates; a cancelled stale re-index puts its candidates back
- configure calls are chained so an older enable cannot land after a
newer disable; a leftover WAL without index.sqlite reports no index;
a fractional historyDays collapses to all history
- coverage polling ignores answers older than the latest request, and a
completed poll outranks the retained result's snapshot
- CLI: --since accepts only ISO 8601, ~user paths are left alone, and
transcript fields are stripped of escape sequences and control bytes
before they reach the terminal; search_log stores the redacted query
- search evidence replaces the latest-turn preview instead of stacking
- Cursor meta index rethrows WslTranscriptFsError instead of caching a
session without its metadata; worker coverage reads forward the signal
- Windows has no load average, so the backfill backs off on its own CPU
share there; '{{count}} matches' pluralizes
This commit is contained in:
@@ -33,7 +33,7 @@ function makeHit(overrides: Partial<AiVaultSearchHit> = {}): AiVaultSearchHit {
|
||||
evidence: {
|
||||
role: 'assistant',
|
||||
timestamp: '2026-09-01T11:29:00.000Z',
|
||||
snippet: 'the [strict] [mode] guard fires\ntwice',
|
||||
snippet: 'the [[strict]] [[mode]] guard fires\ntwice',
|
||||
...overrides.evidence
|
||||
},
|
||||
...overrides
|
||||
@@ -64,6 +64,28 @@ afterEach(() => {
|
||||
})
|
||||
|
||||
describe('formatAgentSessionSearch', () => {
|
||||
it('strips terminal escape sequences and control bytes from transcript fields', () => {
|
||||
const output = format(
|
||||
makeResult({
|
||||
hits: [
|
||||
makeHit({
|
||||
title: 'Fix \u001b]52;c;Zm9v\u0007 clipboard',
|
||||
evidence: {
|
||||
role: 'tool',
|
||||
timestamp: null,
|
||||
snippet: 'ran \u001b[31m[[rm]]\u001b[0m -rf\u0000 /tmp'
|
||||
}
|
||||
})
|
||||
]
|
||||
}),
|
||||
'rm'
|
||||
)
|
||||
expect(output).not.toContain('\u001b')
|
||||
expect(output).not.toContain('\u0000')
|
||||
expect(output).toContain('Fix clipboard')
|
||||
expect(output).toContain('ran [[rm]] -rf /tmp')
|
||||
})
|
||||
|
||||
it('reports no match with the query when there are no hits', () => {
|
||||
const output = format(makeResult({ hits: [] }), 'kernel panic')
|
||||
expect(output.split('\n')[0]).toBe('No sessions match "kernel panic".')
|
||||
@@ -79,7 +101,7 @@ describe('formatAgentSessionSearch', () => {
|
||||
|
||||
it('renders the evidence role glyph and a single-line snippet', () => {
|
||||
const evidence = format(makeResult()).split('\n')[1]
|
||||
expect(evidence).toBe(' agent ▸ the [strict] [mode] guard fires twice')
|
||||
expect(evidence).toBe(' agent ▸ the [[strict]] [[mode]] guard fires twice')
|
||||
})
|
||||
|
||||
it('labels each evidence role distinctly', () => {
|
||||
|
||||
@@ -1,3 +1,7 @@
|
||||
import {
|
||||
stripAnsiEscapeSequences,
|
||||
TERMINAL_CONTROL_CHARACTER_PATTERN
|
||||
} from '../shared/ansi-escape-sequences'
|
||||
import { basename } from 'node:path'
|
||||
import type { AiVaultSearchIndexStatus } from '../shared/ai-vault-search-settings'
|
||||
import { aiVaultAgentLabel } from '../shared/ai-vault-types'
|
||||
@@ -41,13 +45,19 @@ function projectLabel(hit: AiVaultSearchHit): string {
|
||||
return hit.branch ? `${cwd} · ${hit.branch}` : cwd
|
||||
}
|
||||
|
||||
// Why: transcript text reaches the terminal verbatim; an OSC 52 or cursor
|
||||
// sequence inside a tool log would otherwise execute on the user's terminal.
|
||||
function terminalSafe(value: string): string {
|
||||
return stripAnsiEscapeSequences(value).replace(TERMINAL_CONTROL_CHARACTER_PATTERN, '')
|
||||
}
|
||||
|
||||
function formatHit(index: number, hit: AiVaultSearchHit): string {
|
||||
const header = `${String(index + 1).padStart(2)}. ${hit.title}`
|
||||
const meta = `${aiVaultAgentLabel(hit.agent)} · ${projectLabel(hit)} · ${relativeAge(hit.updatedAt)}`
|
||||
const header = `${String(index + 1).padStart(2)}. ${terminalSafe(hit.title)}`
|
||||
const meta = `${aiVaultAgentLabel(hit.agent)} · ${terminalSafe(projectLabel(hit))} · ${relativeAge(hit.updatedAt)}`
|
||||
const evidence = hit.evidence.snippet
|
||||
? ` ${ROLE_LABEL[hit.evidence.role]} ▸ ${hit.evidence.snippet.replaceAll('\n', ' ')}`
|
||||
? ` ${ROLE_LABEL[hit.evidence.role]} ▸ ${terminalSafe(hit.evidence.snippet).replaceAll('\n', ' ')}`
|
||||
: null
|
||||
const resume = ` resume: ${hit.resumeCommand}${hit.cwd ? ` (cwd ${hit.cwd})` : ''}`
|
||||
const resume = ` resume: ${terminalSafe(hit.resumeCommand)}${hit.cwd ? ` (cwd ${terminalSafe(hit.cwd)})` : ''}`
|
||||
return [`${header} ${meta}`, evidence, resume].filter(Boolean).join('\n')
|
||||
}
|
||||
|
||||
|
||||
@@ -135,6 +135,23 @@ describe('orca search --agent-session', () => {
|
||||
expect(callMock).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('rejects non-ISO dates that Date.parse would accept', async () => {
|
||||
for (const since of ['08/01/2026', 'Sat, 01 Aug 2026 00:00:00 GMT']) {
|
||||
const error = await runSearch({ 'agent-session': 'q', since }).catch(
|
||||
(caught: unknown) => caught
|
||||
)
|
||||
expect((error as RuntimeClientError).code).toBe('invalid_argument')
|
||||
}
|
||||
expect(callMock).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('leaves a ~user path alone instead of splicing HOME into it', async () => {
|
||||
vi.stubEnv('HOME', '/home/tester')
|
||||
await runSearch({ 'agent-session': 'q', path: '~other/repo' })
|
||||
expect(searchParams().scopePaths).toEqual(['~other/repo'])
|
||||
vi.unstubAllEnvs()
|
||||
})
|
||||
|
||||
it('normalizes --since to an ISO timestamp', async () => {
|
||||
await runSearch({ 'agent-session': 'q', since: '2026-08-01T00:00:00+02:00' })
|
||||
expect(searchParams().since).toBe('2026-07-31T22:00:00.000Z')
|
||||
|
||||
@@ -35,12 +35,25 @@ function parseAgents(flags: Map<string, string | boolean>): AiVaultAgent[] | und
|
||||
return agents
|
||||
}
|
||||
|
||||
const ISO_8601 =
|
||||
/^\d{4}-\d{2}-\d{2}(?:[T ]\d{2}:\d{2}(?::\d{2}(?:\.\d{1,9})?)?(?:Z|[+-]\d{2}:?\d{2})?)?$/
|
||||
|
||||
/** `~` and `~/x` are the home directory; `~other/x` is left for the host to resolve. */
|
||||
function expandHomePath(value: string): string {
|
||||
const home = process.env.HOME
|
||||
if (!home || (value !== '~' && !value.startsWith('~/'))) {
|
||||
return value
|
||||
}
|
||||
return `${home}${value.slice(1)}`
|
||||
}
|
||||
|
||||
function parseSince(flags: Map<string, string | boolean>): string | undefined {
|
||||
const value = getOptionalStringFlag(flags, 'since')
|
||||
if (value === undefined) {
|
||||
return undefined
|
||||
}
|
||||
const parsed = Date.parse(value)
|
||||
// Why: Date.parse also accepts `08/01/2026` and RFC 2822; the flag documents ISO 8601.
|
||||
const parsed = ISO_8601.test(value) ? Date.parse(value) : Number.NaN
|
||||
if (!Number.isFinite(parsed)) {
|
||||
throw new RuntimeClientError('invalid_argument', '--since must be an ISO 8601 timestamp.')
|
||||
}
|
||||
@@ -78,9 +91,7 @@ export const SEARCH_HANDLERS: Record<string, CommandHandler> = {
|
||||
'Missing --agent-session <query>. Example: orca search --agent-session "strict mode violation"'
|
||||
)
|
||||
}
|
||||
const scopePaths = getRepeatedStringFlag(flags, 'path').map((value) =>
|
||||
value.startsWith('~') ? value.replace(/^~/, process.env.HOME ?? '~') : value
|
||||
)
|
||||
const scopePaths = getRepeatedStringFlag(flags, 'path').map(expandHomePath)
|
||||
const result = await client.call<AiVaultSearchResult>('aiVault.searchSessions', {
|
||||
query,
|
||||
limit: getOptionalPositiveIntegerFlag(flags, 'limit'),
|
||||
|
||||
@@ -2,8 +2,26 @@ import { existsSync } from 'node:fs'
|
||||
import { mkdir, mkdtemp, rm, utimes, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
// Why: the backfill's first await; failing it once is the cheapest way to make
|
||||
// a whole pass throw without touching the transcripts on disk.
|
||||
let failNextParseCacheLoad = false
|
||||
vi.mock('../ai-vault/session-parse-cache-persistence', async (importOriginal) => {
|
||||
const actual = await importOriginal<typeof ParseCachePersistence>()
|
||||
return {
|
||||
...actual,
|
||||
ensureSessionParseCacheLoaded: (): Promise<void> => {
|
||||
if (failNextParseCacheLoad) {
|
||||
failNextParseCacheLoad = false
|
||||
return Promise.reject(new Error('parse cache unavailable'))
|
||||
}
|
||||
return actual.ensureSessionParseCacheLoaded()
|
||||
}
|
||||
}
|
||||
})
|
||||
import { getSessionSearchIndexSink } from '../ai-vault/session-search-capture'
|
||||
import type * as ParseCachePersistence from '../ai-vault/session-parse-cache-persistence'
|
||||
import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache'
|
||||
import { isolatedScanRoots, jsonLines } from '../ai-vault/session-scanner-test-fixtures'
|
||||
import { SessionSearchService, type SessionSearchScanRoots } from './session-search-service'
|
||||
@@ -158,6 +176,35 @@ describe('SessionSearchService consent gate', () => {
|
||||
expect(existsSync(`${databasePath}-wal`)).toBe(false)
|
||||
})
|
||||
|
||||
it('retries a backfill that failed instead of memoizing the failure', async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'retry-session', 'the vacuum quota never settles')
|
||||
|
||||
const service = makeService(databasePath, { enabled: true, historyDays: null })
|
||||
failNextParseCacheLoad = true
|
||||
await service.ensureBackfill(roots)
|
||||
expect(service.coverage().sessionsIndexed).toBe(0)
|
||||
expect(service.coverage().backfill).toBe('idle')
|
||||
|
||||
await service.ensureBackfill(roots)
|
||||
const result = await service.search({ query: 'vacuum', refresh: false }, roots)
|
||||
expect(result.hits.map((hit) => hit.sessionId)).toContain('retry-session')
|
||||
})
|
||||
|
||||
it('re-enumerates when the history bound is widened after a finished backfill', async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'recent-session', 'the vacuum quota never settles', 1)
|
||||
await writeClaudeTranscript(roots, 'ancient-session', 'the vacuum quota never settles', 120)
|
||||
|
||||
const service = makeService(databasePath, { enabled: true, historyDays: 30 })
|
||||
await service.ensureBackfill(roots)
|
||||
await service.configure({ enabled: true, historyDays: null }, roots)
|
||||
await service.ensureBackfill(roots)
|
||||
|
||||
const result = await service.search({ query: 'vacuum', refresh: false }, roots)
|
||||
expect(result.hits.map((hit) => hit.sessionId)).toContain('ancient-session')
|
||||
})
|
||||
|
||||
it('skips transcripts older than the history bound', async () => {
|
||||
const { roots, databasePath } = await scanRoots()
|
||||
await writeClaudeTranscript(roots, 'recent-session', 'the vacuum quota never settles', 1)
|
||||
|
||||
@@ -110,4 +110,13 @@ describe('readAiVaultSearchIndexStatus', () => {
|
||||
|
||||
expect(readAiVaultSearchIndexStatus().indexSizeBytes).toBe(123)
|
||||
})
|
||||
|
||||
it('treats a leftover sidecar without the database as no index', async () => {
|
||||
const userData = await makeUserDataDir()
|
||||
initSessionSearchPaths(userData)
|
||||
await mkdir(join(userData, 'ai-vault-search'), { recursive: true })
|
||||
await writeFile(join(userData, 'ai-vault-search', 'index.sqlite-wal'), 'y'.repeat(23))
|
||||
|
||||
expect(readAiVaultSearchIndexStatus().indexSizeBytes).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -35,9 +35,19 @@ export function applyAiVaultSearchSettings(
|
||||
return Promise.resolve(null)
|
||||
}
|
||||
const policy: AiVaultSearchSettings = resolveAiVaultSearchSettings(settings)
|
||||
return configureAiVaultSearch({ databasePath, ...policy }, { clearIndex: options.clearIndex })
|
||||
// Why: each call resolves scan roots before it reaches the scanner, so two
|
||||
// overlapping toggles could land out of order; apply them one after another.
|
||||
const applied = applyChain
|
||||
.catch(() => undefined)
|
||||
.then(() =>
|
||||
configureAiVaultSearch({ databasePath, ...policy }, { clearIndex: options.clearIndex })
|
||||
)
|
||||
applyChain = applied
|
||||
return applied
|
||||
}
|
||||
|
||||
let applyChain: Promise<unknown> = Promise.resolve()
|
||||
|
||||
export function readAiVaultSearchIndexStatus(): AiVaultSearchIndexStatus {
|
||||
return { ...getSessionSearchPolicy(), indexSizeBytes: readAiVaultSearchIndexSizeBytes() }
|
||||
}
|
||||
@@ -76,12 +86,18 @@ export function readAiVaultSearchIndexSizeBytes(): number | null {
|
||||
if (!databasePath) {
|
||||
return null
|
||||
}
|
||||
let total: number | null = null
|
||||
for (const suffix of ['', '-wal', '-shm', '-journal']) {
|
||||
let total: number
|
||||
try {
|
||||
total = statSync(databasePath).size
|
||||
} catch {
|
||||
// Why: a leftover sidecar without the main file is not an index.
|
||||
return null
|
||||
}
|
||||
for (const suffix of ['-wal', '-shm', '-journal']) {
|
||||
try {
|
||||
total = (total ?? 0) + statSync(`${databasePath}${suffix}`).size
|
||||
total += statSync(`${databasePath}${suffix}`).size
|
||||
} catch {
|
||||
// A missing sidecar is normal; only a missing main file means "no index".
|
||||
// A missing sidecar is normal.
|
||||
}
|
||||
}
|
||||
return total
|
||||
|
||||
@@ -6,7 +6,9 @@ import type {
|
||||
} from '../../shared/ai-vault-search-types'
|
||||
import {
|
||||
AI_VAULT_SEARCH_LIMIT_DEFAULT,
|
||||
AI_VAULT_SEARCH_LIMIT_MAX
|
||||
AI_VAULT_SEARCH_LIMIT_MAX,
|
||||
AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE,
|
||||
AI_VAULT_SEARCH_SNIPPET_MARK_OPEN
|
||||
} from '../../shared/ai-vault-search-types'
|
||||
import type { AiVaultAgent } from '../../shared/ai-vault-types'
|
||||
import {
|
||||
@@ -32,6 +34,10 @@ const MESSAGE_CANDIDATE_LIMIT = 600
|
||||
// Subtracted per session: `0.02 · ln(1 + messages)`; slightly positive on both eval sets.
|
||||
const LENGTH_PRIOR = 0.02
|
||||
const SNIPPET_TOKENS = 12
|
||||
// Why: single brackets are everywhere in code transcripts (`arr[0]`, regex
|
||||
// classes, markdown links) and would read as matches; doubled ones are rare.
|
||||
const SNIPPET_MARK_OPEN = AI_VAULT_SEARCH_SNIPPET_MARK_OPEN
|
||||
const SNIPPET_MARK_CLOSE = AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE
|
||||
|
||||
type MessageRow = {
|
||||
rowid: number
|
||||
@@ -103,7 +109,7 @@ export class SessionSearchQuery {
|
||||
}
|
||||
const exact = this.retrieveLiteral(plan, retrieval.tier)
|
||||
if (exact) {
|
||||
return { hits: this.rollUp(exact.rows, retrieval), route: exact.route }
|
||||
return { hits: this.rollUp(exact.rows, retrieval, plan), route: exact.route }
|
||||
}
|
||||
// Why: repair runs before the OR fallback, not after it fails; a typo next
|
||||
// to a common word would otherwise be masked by the common word's hits.
|
||||
@@ -115,7 +121,7 @@ export class SessionSearchQuery {
|
||||
route: 'or' as const
|
||||
}
|
||||
return {
|
||||
hits: this.rollUp(result.rows, retrieval),
|
||||
hits: this.rollUp(result.rows, retrieval, effective),
|
||||
route: repaired ? (`typo+${result.route}` as AiVaultSearchRoute) : result.route,
|
||||
...(repaired ? { repairedTerms: repaired.body } : {})
|
||||
}
|
||||
@@ -192,7 +198,11 @@ export class SessionSearchQuery {
|
||||
}
|
||||
}
|
||||
|
||||
private rollUp(rows: MessageRow[], retrieval: Retrieval): AiVaultSearchHit[] {
|
||||
private rollUp(
|
||||
rows: MessageRow[],
|
||||
retrieval: Retrieval,
|
||||
plan: SessionSearchQueryPlan
|
||||
): AiVaultSearchHit[] {
|
||||
const best = new Map<number, MessageRow>()
|
||||
for (const row of rows) {
|
||||
const current = best.get(row.session_row_id)
|
||||
@@ -230,20 +240,22 @@ export class SessionSearchQuery {
|
||||
evidence: {
|
||||
role: message.role as AiVaultSearchHit['evidence']['role'],
|
||||
timestamp: message.ts,
|
||||
snippet: this.snippet(table, message.rowid, retrieval.text)
|
||||
snippet: this.snippet(table, message.rowid, plan)
|
||||
}
|
||||
}))
|
||||
}
|
||||
|
||||
private snippet(table: string, rowid: number, query: string): string {
|
||||
const expression = orExpression(planSessionSearchQuery(query).terms)
|
||||
// Why: the snippet must highlight the terms that actually retrieved the row,
|
||||
// so a hit found through typo repair is marked with the repaired terms.
|
||||
private snippet(table: string, rowid: number, plan: SessionSearchQueryPlan): string {
|
||||
const expression = orExpression(plan.terms)
|
||||
try {
|
||||
// Why: a bound `rowid = ?` or `rowid IN (?)` next to MATCH is silently
|
||||
// ignored by the FTS5 planner (it returns the first match); only the
|
||||
// subselect form is honoured. Column -1 picks whichever column matched.
|
||||
const row = this.db
|
||||
.prepare(
|
||||
`SELECT snippet(${table}, -1, '[', ']', '…', ${SNIPPET_TOKENS}) AS s
|
||||
`SELECT snippet(${table}, -1, '${SNIPPET_MARK_OPEN}', '${SNIPPET_MARK_CLOSE}', '…', ${SNIPPET_TOKENS}) AS s
|
||||
FROM ${table} WHERE ${table} MATCH ? AND rowid IN (SELECT ?)`
|
||||
)
|
||||
.get(expression, rowid) as { s: string } | undefined
|
||||
@@ -261,8 +273,13 @@ export class SessionSearchQuery {
|
||||
}
|
||||
}
|
||||
|
||||
// Why: the desktop IPC forwards its payload unvalidated, so a non-positive
|
||||
// limit must be clamped here or `LIMIT -1` / `slice(0, -1)` leak through.
|
||||
function resolveLimit(args: AiVaultSearchArgs): number {
|
||||
return Math.min(args.limit ?? AI_VAULT_SEARCH_LIMIT_DEFAULT, AI_VAULT_SEARCH_LIMIT_MAX)
|
||||
const requested = Number.isInteger(args.limit)
|
||||
? (args.limit as number)
|
||||
: AI_VAULT_SEARCH_LIMIT_DEFAULT
|
||||
return Math.min(Math.max(1, requested), AI_VAULT_SEARCH_LIMIT_MAX)
|
||||
}
|
||||
|
||||
function sessionFields(session: SessionRow): Omit<AiVaultSearchHit, 'score' | 'evidence'> {
|
||||
|
||||
@@ -29,10 +29,14 @@ export function sessionRowFilter(
|
||||
filter.values.push(args.since)
|
||||
}
|
||||
if (args.scopePaths && args.scopePaths.length > 0) {
|
||||
filter.conditions.push(`(${args.scopePaths.map(() => '(cwd = ? OR cwd LIKE ?)').join(' OR ')})`)
|
||||
filter.conditions.push(
|
||||
`(${args.scopePaths.map(() => `(${CWD} = ? OR ${CWD} LIKE ? ESCAPE '\\')`).join(' OR ')})`
|
||||
)
|
||||
for (const scope of args.scopePaths) {
|
||||
const trimmed = scope.replace(/[\\/]+$/, '')
|
||||
filter.values.push(trimmed, `${trimmed}${trimmed.includes('\\') ? '\\' : '/'}%`)
|
||||
// Same spelling as CWD so a Windows scope matches a POSIX-recorded cwd,
|
||||
// and escaped so `%`/`_` in a folder name cannot widen the match.
|
||||
const normalized = scope.replaceAll('\\', '/').replace(/\/+$/, '')
|
||||
filter.values.push(normalized, `${escapeLike(normalized)}/%`)
|
||||
}
|
||||
}
|
||||
// Operators narrow, never widen: each one is its own ANDed condition on top
|
||||
|
||||
@@ -45,7 +45,38 @@ const REFRESH_RECENT_PER_AGENT = 12
|
||||
const BACKFILL_LOAD_PER_CPU_CEILING = 1.5
|
||||
const BACKFILL_LOAD_PAUSE_MS = 15_000
|
||||
|
||||
/** null (all history) is the widest bound; otherwise more days means wider. */
|
||||
function widensHistory(previous: number | null, next: number | null): boolean {
|
||||
if (previous === null) {
|
||||
return false
|
||||
}
|
||||
return next === null || next > previous
|
||||
}
|
||||
|
||||
// Why: Windows reports a zero load average, so the only signal left is the
|
||||
// scanner's own recent CPU share; back off when it has been near a full core.
|
||||
const WINDOWS_SELF_CPU_SHARE_CEILING = 0.8
|
||||
let lastCpuSample: { usage: NodeJS.CpuUsage; at: number } | null = null
|
||||
|
||||
function selfCpuShareSinceLastYield(): number {
|
||||
const now = performance.now()
|
||||
const usage = process.cpuUsage()
|
||||
const previous = lastCpuSample
|
||||
lastCpuSample = { usage, at: now }
|
||||
if (!previous) {
|
||||
return 0
|
||||
}
|
||||
const busyMs = (usage.user - previous.usage.user + usage.system - previous.usage.system) / 1000
|
||||
const elapsedMs = Math.max(1, now - previous.at)
|
||||
return busyMs / elapsedMs
|
||||
}
|
||||
|
||||
function backfillPauseMs(): number {
|
||||
if (process.platform === 'win32') {
|
||||
return selfCpuShareSinceLastYield() > WINDOWS_SELF_CPU_SHARE_CEILING
|
||||
? BACKFILL_LOAD_PAUSE_MS
|
||||
: BACKFILL_YIELD_MS
|
||||
}
|
||||
const perCpu = loadavg()[0] / Math.max(1, cpus().length)
|
||||
return perCpu > BACKFILL_LOAD_PER_CPU_CEILING ? BACKFILL_LOAD_PAUSE_MS : BACKFILL_YIELD_MS
|
||||
}
|
||||
@@ -122,9 +153,14 @@ export class SessionSearchService {
|
||||
options: { clearIndex?: boolean } = {}
|
||||
): Promise<AiVaultSearchCoverage> {
|
||||
const wasEnabled = this.policy.enabled
|
||||
const previousDays = this.policy.historyDays
|
||||
this.policy = { enabled: next.enabled, historyDays: next.historyDays }
|
||||
if (options.clearIndex || (wasEnabled && !next.enabled)) {
|
||||
await this.stop()
|
||||
} else if (wasEnabled && widensHistory(previousDays, next.historyDays)) {
|
||||
// Why: a finished backfill is memoized; a wider window has files it
|
||||
// skipped, so it has to enumerate again (rows already indexed are reused).
|
||||
await this.stop({ keepStore: true })
|
||||
}
|
||||
if (options.clearIndex) {
|
||||
removeSessionSearchDatabase(this.databasePath)
|
||||
@@ -146,11 +182,19 @@ export class SessionSearchService {
|
||||
}
|
||||
if (!this.backfillRun) {
|
||||
this.backfillController = new AbortController()
|
||||
this.backfillRun = this.runBackfill(roots, this.backfillController.signal)
|
||||
.catch((error) => console.warn('[ai-vault-search] backfill stopped:', error))
|
||||
const run = this.runBackfill(roots, this.backfillController.signal)
|
||||
.catch((error) => {
|
||||
console.warn('[ai-vault-search] backfill stopped:', error)
|
||||
// Why: a failed pass must not be memoized as done; the next search
|
||||
// gets to try again instead of reporting an incomplete index forever.
|
||||
if (this.backfillRun === run) {
|
||||
this.backfillRun = null
|
||||
}
|
||||
})
|
||||
.finally(() => {
|
||||
this.backfillController = null
|
||||
})
|
||||
this.backfillRun = run
|
||||
}
|
||||
return this.backfillRun
|
||||
}
|
||||
@@ -173,14 +217,16 @@ export class SessionSearchService {
|
||||
}
|
||||
|
||||
/** Waits for the aborted backfill so its last parse cannot write to a closed store. */
|
||||
private async stop(): Promise<void> {
|
||||
private async stop(options: { keepStore?: boolean } = {}): Promise<void> {
|
||||
this.backfillController?.abort()
|
||||
const run = this.backfillRun
|
||||
this.backfillRun = null
|
||||
if (run) {
|
||||
await run.catch(() => undefined)
|
||||
}
|
||||
this.closeStore()
|
||||
if (!options.keepStore) {
|
||||
this.closeStore()
|
||||
}
|
||||
}
|
||||
|
||||
private closeStore(): void {
|
||||
@@ -206,11 +252,21 @@ export class SessionSearchService {
|
||||
}
|
||||
|
||||
private async reindexStale(signal?: AbortSignal): Promise<void> {
|
||||
const stale = this.store?.takeStale() ?? []
|
||||
if (stale.length === 0) {
|
||||
const store = this.store
|
||||
const stale = store?.takeStale() ?? []
|
||||
if (stale.length === 0 || !store) {
|
||||
return
|
||||
}
|
||||
await this.parseAll(stale, signal)
|
||||
try {
|
||||
await this.parseAll(stale, signal)
|
||||
} catch (error) {
|
||||
// Why: a cancelled search (the renderer retires them per keystroke) must
|
||||
// not lose the queue; whatever did not get parsed goes back for next time.
|
||||
for (const candidate of stale) {
|
||||
store.markStale(candidate)
|
||||
}
|
||||
throw error
|
||||
}
|
||||
}
|
||||
|
||||
/** The retention bound is enforced on enumeration; already-indexed rows survive until a clear. */
|
||||
|
||||
@@ -78,7 +78,7 @@ describe('SessionSearchStore', () => {
|
||||
cwd: '/repo/app',
|
||||
evidence: { role: 'tool' }
|
||||
})
|
||||
expect(literal.hits[0]?.evidence.snippet).toContain('[strict]')
|
||||
expect(literal.hits[0]?.evidence.snippet).toContain('[[strict]]')
|
||||
expect(literal.route).toBe('phrase')
|
||||
|
||||
// Identifier split: a partial camelCase name still matches.
|
||||
@@ -202,6 +202,12 @@ describe('SessionSearchStore', () => {
|
||||
expect(store.search({ query: 'shared phrase', agents: ['codex'] }).hits).toHaveLength(0)
|
||||
expect(store.search({ query: 'shared phrase', scopePaths: ['/repo'] }).hits).toHaveLength(2)
|
||||
expect(store.search({ query: 'shared phrase', scopePaths: ['/other'] }).hits).toHaveLength(0)
|
||||
// `_` is a LIKE wildcard; an escaped scope must not match `/repo` through `/r_po`.
|
||||
expect(store.search({ query: 'shared phrase', scopePaths: ['/r_po'] }).hits).toHaveLength(0)
|
||||
expect(store.search({ query: 'shared phrase', scopePaths: ['\\repo\\'] }).hits).toHaveLength(2)
|
||||
// A non-positive limit from an unvalidated caller is clamped, not passed to SQL.
|
||||
expect(store.search({ query: 'shared phrase', limit: -1 }).hits).toHaveLength(1)
|
||||
expect(store.search({ query: 'shared phrase', limit: 0 }).hits).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('keeps a provider with discovered files and no indexed sessions in coverage', async () => {
|
||||
|
||||
@@ -13,6 +13,7 @@ import type {
|
||||
SessionSearchIndexUpdate
|
||||
} from '../ai-vault/session-search-capture'
|
||||
import type { SessionFileCandidate } from '../ai-vault/session-scanner-types'
|
||||
import { redactSessionSearchText } from './session-search-redaction'
|
||||
import { SessionSearchIndexWriter } from './session-search-index-writer'
|
||||
import { SessionSearchQuery } from './session-search-query'
|
||||
import { openSessionSearchDatabase } from './session-search-schema'
|
||||
@@ -177,7 +178,7 @@ export class SessionSearchStore implements SessionSearchIndexSink {
|
||||
.prepare(
|
||||
'INSERT INTO search_log(ts, query, route, hits, duration_ms) VALUES (?, ?, ?, ?, ?)'
|
||||
)
|
||||
.run(new Date().toISOString(), query, route, hits, durationMs)
|
||||
.run(new Date().toISOString(), redactSessionSearchText(query), route, hits, durationMs)
|
||||
this.db
|
||||
.prepare(
|
||||
`DELETE FROM search_log WHERE id <= (
|
||||
|
||||
@@ -104,7 +104,7 @@ export function readAiVaultSearchCoverageInBackground(
|
||||
): Promise<AiVaultSearchCoverage> {
|
||||
return shouldUseAiVaultServiceProcess()
|
||||
? readAiVaultSearchCoverageInService(request, signal)
|
||||
: readAiVaultSearchCoverageInWorker(request)
|
||||
: readAiVaultSearchCoverageInWorker(request, signal)
|
||||
}
|
||||
|
||||
/** Null when no scanner is running: the next spawn reads the policy from its init payload. */
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { basename, dirname, join } from 'node:path'
|
||||
import { wslGatedReaddir, wslGatedStat } from '../native-chat/wsl-transcript-fs-access'
|
||||
import { WslTranscriptFsError } from '../native-chat/wsl-transcript-fs-gate'
|
||||
import { timestampIso } from './session-scanner-accumulator'
|
||||
import { extractString, normalizeTitleText, readJsonObjectIfExists } from './session-scanner-values'
|
||||
|
||||
@@ -86,7 +87,12 @@ async function readCursorChatMetaIndex(chatsRoot: string): Promise<Map<string, s
|
||||
.filter((entry) => entry.isDirectory())
|
||||
.map((entry) => entry.name)
|
||||
.sort()
|
||||
} catch {
|
||||
} catch (error) {
|
||||
// Why: a refused WSL read is not "no chats"; letting it through keeps the
|
||||
// session out of the parse cache instead of caching it without metadata.
|
||||
if (error instanceof WslTranscriptFsError) {
|
||||
throw error
|
||||
}
|
||||
return new Map()
|
||||
}
|
||||
const signature = await readCursorChatsSignature(chatsRoot, workspaceDirs)
|
||||
@@ -130,7 +136,10 @@ async function buildCursorChatMetaIndex(
|
||||
let chatDirs
|
||||
try {
|
||||
chatDirs = await wslGatedReaddir(join(chatsRoot, workspaceDir), 'scan')
|
||||
} catch {
|
||||
} catch (error) {
|
||||
if (error instanceof WslTranscriptFsError) {
|
||||
throw error
|
||||
}
|
||||
continue
|
||||
}
|
||||
for (const chatDir of chatDirs) {
|
||||
|
||||
@@ -82,11 +82,13 @@ export class AiVaultScannerWorkerClient {
|
||||
}
|
||||
|
||||
searchCoverage(
|
||||
request: Pick<AiVaultServiceSearchRequest, 'roots'>
|
||||
request: Pick<AiVaultServiceSearchRequest, 'roots'>,
|
||||
signal?: AbortSignal
|
||||
): Promise<AiVaultSearchCoverage> {
|
||||
return this.dispatch(
|
||||
{ kind: 'searchCoverage', request },
|
||||
TITLE_TIMEOUT_MS
|
||||
TITLE_TIMEOUT_MS,
|
||||
signal
|
||||
) as Promise<AiVaultSearchCoverage>
|
||||
}
|
||||
|
||||
|
||||
@@ -66,9 +66,10 @@ export function searchAiVaultSessionsInWorker(
|
||||
}
|
||||
|
||||
export function readAiVaultSearchCoverageInWorker(
|
||||
request: Pick<AiVaultServiceSearchRequest, 'roots'>
|
||||
request: Pick<AiVaultServiceSearchRequest, 'roots'>,
|
||||
signal?: AbortSignal
|
||||
): Promise<AiVaultSearchCoverage> {
|
||||
return getSharedClient().searchCoverage(request)
|
||||
return getSharedClient().searchCoverage(request, signal)
|
||||
}
|
||||
|
||||
/** Null when no worker exists yet; the next one reads the new policy from its workerData. */
|
||||
|
||||
@@ -8,7 +8,7 @@ afterEach(cleanup)
|
||||
|
||||
describe('aiVaultSearchSnippetSegments', () => {
|
||||
it('splits a snippet on its match markers', () => {
|
||||
expect(aiVaultSearchSnippetSegments('fix the [strict] mode [bug]')).toEqual([
|
||||
expect(aiVaultSearchSnippetSegments('fix the [[strict]] mode [[bug]]')).toEqual([
|
||||
{ text: 'fix the ', matched: false },
|
||||
{ text: 'strict', matched: true },
|
||||
{ text: ' mode ', matched: false },
|
||||
@@ -16,6 +16,14 @@ describe('aiVaultSearchSnippetSegments', () => {
|
||||
])
|
||||
})
|
||||
|
||||
it('keeps literal brackets from a transcript as plain text', () => {
|
||||
expect(aiVaultSearchSnippetSegments('read arr[0] then [[index]] it')).toEqual([
|
||||
{ text: 'read arr[0] then ', matched: false },
|
||||
{ text: 'index', matched: true },
|
||||
{ text: ' it', matched: false }
|
||||
])
|
||||
})
|
||||
|
||||
it('leaves an unmarked snippet as one plain segment', () => {
|
||||
expect(aiVaultSearchSnippetSegments('no markers here')).toEqual([
|
||||
{ text: 'no markers here', matched: false }
|
||||
@@ -27,7 +35,7 @@ describe('AiVaultSearchEvidenceLine', () => {
|
||||
it('renders the role and marks the matched terms', () => {
|
||||
render(
|
||||
<AiVaultSearchEvidenceLine
|
||||
evidence={{ role: 'user', timestamp: null, snippet: 'fix the [strict] mode' }}
|
||||
evidence={{ role: 'user', timestamp: null, snippet: 'fix the [[strict]] mode' }}
|
||||
/>
|
||||
)
|
||||
expect(screen.getByText('You')).toBeTruthy()
|
||||
|
||||
@@ -12,11 +12,11 @@ const ROLE_GLYPH = {
|
||||
|
||||
export type AiVaultSearchSnippetSegment = { text: string; matched: boolean }
|
||||
|
||||
/** Splits an FTS5 snippet on its `[term]` match markers. */
|
||||
/** Splits an FTS5 snippet on its `[[term]]` match markers; single brackets are text. */
|
||||
export function aiVaultSearchSnippetSegments(snippet: string): AiVaultSearchSnippetSegment[] {
|
||||
const segments: AiVaultSearchSnippetSegment[] = []
|
||||
let cursor = 0
|
||||
const pattern = /\[([^[\]]*)\]/g
|
||||
const pattern = /\[\[(.*?)\]\]/g
|
||||
let match: RegExpExecArray | null
|
||||
while ((match = pattern.exec(snippet)) !== null) {
|
||||
if (match.index > cursor) {
|
||||
|
||||
@@ -188,7 +188,7 @@ export function VaultSessionRow({
|
||||
onRequestDelete={requestDelete}
|
||||
/>
|
||||
</div>
|
||||
{!detailsExpanded && (latestTurn || !searchEvidence) ? (
|
||||
{!detailsExpanded && !searchEvidence ? (
|
||||
<div className="mt-0.5 min-w-0 line-clamp-2 text-[12px] leading-4 text-muted-foreground">
|
||||
{latestTurn ? (
|
||||
<>
|
||||
|
||||
@@ -59,6 +59,27 @@ describe('useAiVaultSearchCoveragePoll', () => {
|
||||
expect(searchCoverage).toHaveBeenCalledTimes(callsAfterFirstRead)
|
||||
})
|
||||
|
||||
it('ignores a slow running answer that lands after a newer complete one', async () => {
|
||||
let releaseFirst: (value: ReturnType<typeof coverage>) => void = () => undefined
|
||||
searchCoverage
|
||||
.mockImplementationOnce(
|
||||
() =>
|
||||
new Promise((resolve) => {
|
||||
releaseFirst = resolve
|
||||
})
|
||||
)
|
||||
.mockResolvedValueOnce(coverage('complete'))
|
||||
const { result } = renderHook(() => useAiVaultSearchCoveragePoll(true), { wrapper })
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(AI_VAULT_SEARCH_COVERAGE_POLL_MS)
|
||||
})
|
||||
expect(result.current?.backfill).toBe('complete')
|
||||
await act(async () => {
|
||||
releaseFirst(coverage('running'))
|
||||
})
|
||||
expect(result.current?.backfill).toBe('complete')
|
||||
})
|
||||
|
||||
it('keeps polling while the backfill is still running', async () => {
|
||||
searchCoverage.mockResolvedValue(coverage('running'))
|
||||
renderHook(() => useAiVaultSearchCoveragePoll(true), { wrapper })
|
||||
|
||||
@@ -20,16 +20,20 @@ export function useAiVaultSearchCoveragePoll(enabled: boolean): AiVaultSearchCov
|
||||
return
|
||||
}
|
||||
let stopped = false
|
||||
let generation = 0
|
||||
// One interval, cleared unconditionally on unmount; the reads it drives stop
|
||||
// once the backfill is complete because there is nothing left to watch.
|
||||
const interval = setInterval(() => {
|
||||
read()
|
||||
}, AI_VAULT_SEARCH_COVERAGE_POLL_MS)
|
||||
function read(): void {
|
||||
generation += 1
|
||||
const issued = generation
|
||||
void window.api.aiVault
|
||||
.searchCoverage()
|
||||
.then((next) => {
|
||||
if (stopped) {
|
||||
// Why: a slow 'running' answer must not land after a newer 'complete'.
|
||||
if (stopped || issued !== generation) {
|
||||
return
|
||||
}
|
||||
setCoverage(next)
|
||||
|
||||
@@ -46,11 +46,8 @@ export function useAiVaultSessionSearchRequest(
|
||||
// Serialized so a fresh object identity with identical values does not
|
||||
// restart the debounce on every parent render.
|
||||
const argsKey = args ? JSON.stringify(args) : ''
|
||||
const argsKeyRef = useRef(argsKey)
|
||||
argsKeyRef.current = argsKey
|
||||
|
||||
const issue = useCallback((overrides: Partial<AiVaultSearchArgs>): void => {
|
||||
const key = argsKeyRef.current
|
||||
const issue = useCallback((key: string, overrides: Partial<AiVaultSearchArgs>): void => {
|
||||
if (!key) {
|
||||
return
|
||||
}
|
||||
@@ -75,27 +72,44 @@ export function useAiVaultSessionSearchRequest(
|
||||
})
|
||||
}, [])
|
||||
|
||||
// Why: a flush has to cancel the debounce the args effect armed, and the two
|
||||
// effects cannot share locals, so the args effect publishes its cancel.
|
||||
const cancelDebounceRef = useRef<() => void>(() => undefined)
|
||||
|
||||
useEffect(() => {
|
||||
if (!argsKey) {
|
||||
return
|
||||
}
|
||||
const typingTimer = setTimeout(
|
||||
() => issue({ tier: 'conversation', refresh: false }),
|
||||
() => issue(argsKey, { tier: 'conversation', refresh: false }),
|
||||
AI_VAULT_SEARCH_TYPING_DELAY_MS
|
||||
)
|
||||
const settledTimer = setTimeout(() => issue({ tier: 'full' }), AI_VAULT_SEARCH_SETTLED_DELAY_MS)
|
||||
return () => {
|
||||
const settledTimer = setTimeout(
|
||||
() => issue(argsKey, { tier: 'full' }),
|
||||
AI_VAULT_SEARCH_SETTLED_DELAY_MS
|
||||
)
|
||||
const cancel = (): void => {
|
||||
clearTimeout(typingTimer)
|
||||
clearTimeout(settledTimer)
|
||||
}
|
||||
cancelDebounceRef.current = cancel
|
||||
return () => {
|
||||
cancel()
|
||||
cancelDebounceRef.current = () => undefined
|
||||
// Retires anything still in flight for the query being replaced.
|
||||
sequenceRef.current += 1
|
||||
}
|
||||
}, [argsKey, issue])
|
||||
|
||||
// Why: the args effect must not re-run on flush (it would restart the
|
||||
// debounce), so the flush reads the key it was rendered with. The pending
|
||||
// debounce is dropped so a late typing-tier request cannot outrank it.
|
||||
useEffect(() => {
|
||||
if (flushSignal > 0) {
|
||||
issue({ tier: 'full' })
|
||||
cancelDebounceRef.current()
|
||||
issue(argsKey, { tier: 'full' })
|
||||
}
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [flushSignal, issue])
|
||||
|
||||
const current = settled?.key === argsKey ? settled : null
|
||||
|
||||
@@ -103,8 +103,8 @@ export function useAiVaultSessionSearchResults(input: {
|
||||
key: 'ai-vault-search-results',
|
||||
label: translate(
|
||||
'auto.components.right.sidebar.AiVaultPanel.searchMatches',
|
||||
'{{value0}} matches',
|
||||
{ value0: hitSessions.sessions.length }
|
||||
'{{count}} matches',
|
||||
{ count: hitSessions.sessions.length }
|
||||
),
|
||||
sessions: hitSessions.sessions
|
||||
}
|
||||
@@ -117,7 +117,12 @@ export function useAiVaultSessionSearchResults(input: {
|
||||
loading,
|
||||
updating,
|
||||
error,
|
||||
coverage: result?.coverage ?? polledCoverage,
|
||||
// Why: the poll keeps going after a retained result, so it is the fresher
|
||||
// read once it says the backfill is done.
|
||||
coverage:
|
||||
polledCoverage?.backfill === 'complete'
|
||||
? polledCoverage
|
||||
: (result?.coverage ?? polledCoverage),
|
||||
disabled: isAiVaultSearchDisabled(result?.coverage),
|
||||
flush,
|
||||
groups,
|
||||
|
||||
+3
-1
@@ -687,7 +687,9 @@
|
||||
"localSessionSshWorkspaceUnsupported": "This session's history is stored on this machine, so it can't resume in an SSH workspace. Open a local workspace instead.",
|
||||
"localWorkspacesOnly": "Resume from history is only available in local workspaces.",
|
||||
"remoteBrowseLocalHistory": "Remote workspaces can browse local history. Resume actions run from local workspaces.",
|
||||
"valueCopyFailed": "Unable to copy {{value0}}"
|
||||
"valueCopyFailed": "Unable to copy {{value0}}",
|
||||
"searchMatches_one": "{{count}} match",
|
||||
"searchMatches_other": "{{count}} matches"
|
||||
},
|
||||
"AiVaultPanelControls": {
|
||||
"allSessions": "All sessions",
|
||||
|
||||
@@ -13025,7 +13025,9 @@
|
||||
"prepareSessionResumeFailed": "Could not prepare this session for resume.",
|
||||
"sessionDeleted": "Session deleted",
|
||||
"sessionDeleteFailed": "Couldn't delete the session",
|
||||
"searchMatches": "{{value0}} matches"
|
||||
"searchMatches_one": "{{count}} match",
|
||||
"searchMatches_other": "{{count}} matches",
|
||||
"searchMatches": "{{count}} matches"
|
||||
},
|
||||
"AiVaultPanelControls": {
|
||||
"scanningSessions": "Scanning sessions",
|
||||
|
||||
@@ -25,7 +25,10 @@ export function normalizeAiVaultSearchHistoryDays(value: unknown): number | null
|
||||
if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) {
|
||||
return null
|
||||
}
|
||||
return Math.min(HISTORY_DAYS_MAX, Math.floor(value))
|
||||
const days = Math.floor(value)
|
||||
// Why: a fractional day floors to 0, which the UI shows as "all history"
|
||||
// while the cutoff would be "now"; make the two agree.
|
||||
return days <= 0 ? null : Math.min(HISTORY_DAYS_MAX, days)
|
||||
}
|
||||
|
||||
export function resolveAiVaultSearchSettings(
|
||||
|
||||
@@ -24,7 +24,7 @@ export type AiVaultSearchArgs = {
|
||||
export type AiVaultSearchEvidence = {
|
||||
role: AiVaultSessionPreviewMessage['role']
|
||||
timestamp: string | null
|
||||
/** FTS5 snippet with the matched terms wrapped in `[` `]`. */
|
||||
/** FTS5 snippet with the matched terms wrapped in `[[` `]]`. */
|
||||
snippet: string
|
||||
}
|
||||
|
||||
@@ -84,3 +84,7 @@ export type AiVaultSearchCoverage = {
|
||||
filesPending: number
|
||||
lastIndexedAt: string | null
|
||||
}
|
||||
|
||||
/** Snippet match markers; doubled so literal brackets in code never read as a match. */
|
||||
export const AI_VAULT_SEARCH_SNIPPET_MARK_OPEN = '[['
|
||||
export const AI_VAULT_SEARCH_SNIPPET_MARK_CLOSE = ']]'
|
||||
|
||||
Reference in New Issue
Block a user