diff --git a/src/main/codex-usage/codex-usage-event-attribution.ts b/src/main/codex-usage/codex-usage-event-attribution.ts index 798183360be..4cde7a74af8 100644 --- a/src/main/codex-usage/codex-usage-event-attribution.ts +++ b/src/main/codex-usage/codex-usage-event-attribution.ts @@ -1,11 +1,6 @@ -import { win32, posix } from 'node:path' -import { areWorktreePathsEqual } from '../ipc/worktree-logic' -import { - looksLikeWindowsPath, - normalizeComparablePath, - normalizeFsPath -} from '../usage/usage-path-comparison' +import { normalizeComparablePath } from '../usage/usage-path-comparison' import type { UsageScanWorktreeRef } from '../usage/usage-provider-contract' +import type { UsageWorktreeResolver } from '../usage/usage-worktree-resolver' import type { CodexUsageAttributedEvent, CodexUsageParsedEvent } from './types' export type CodexUsageWorktreeRef = UsageScanWorktreeRef @@ -32,50 +27,9 @@ function localDayFromTimestamp(timestamp: string): string | null { return `${year}-${month}-${day}` } -function isContainingPath(candidatePath: string, targetPath: string): boolean { - const useWin32 = looksLikeWindowsPath(candidatePath) || looksLikeWindowsPath(targetPath) - const relativePath = useWin32 - ? win32.relative(candidatePath, targetPath) - : posix.relative(candidatePath, targetPath) - if (!relativePath) { - return true - } - // Why: on Windows, `path.relative('C:\\repo', 'D:\\other')` returns an - // absolute `D:\\other` path instead of a `..`-prefixed relative. Treating - // that as "contained" would attribute off-drive Codex usage to the wrong - // Orca worktree. - const isAbsoluteRelative = useWin32 - ? win32.isAbsolute(relativePath) - : posix.isAbsolute(relativePath) - const parentPrefix = useWin32 ? `..${win32.sep}` : `..${posix.sep}` - // Why: `..name` is a valid child path; only `..` and `../...` escape. - return ( - !isAbsoluteRelative && - relativePath !== '..' && - !relativePath.startsWith(parentPrefix) && - relativePath !== '.' - ) -} - -function findContainingWorktree( - cwd: string, - worktrees: (CodexUsageWorktreeRef & { canonicalPath: string })[] -): CodexUsageWorktreeRef | null { - const normalizedCwd = normalizeFsPath(cwd) - for (const worktree of worktrees) { - if (areWorktreePathsEqual(worktree.canonicalPath, normalizedCwd)) { - return worktree - } - if (isContainingPath(worktree.canonicalPath, normalizedCwd)) { - return worktree - } - } - return null -} - export async function attributeCodexUsageEvent( event: CodexUsageParsedEvent, - worktrees: (CodexUsageWorktreeRef & { canonicalPath: string })[] + resolveWorktree: UsageWorktreeResolver ): Promise { const day = localDayFromTimestamp(event.timestamp) if (!day) { @@ -88,7 +42,7 @@ export async function attributeCodexUsageEvent( let projectLabel = getDefaultProjectLabel(event.cwd) if (event.cwd) { - const worktree = findContainingWorktree(event.cwd, worktrees) + const worktree = resolveWorktree(event.cwd) if (worktree) { repoId = worktree.repoId worktreeId = worktree.worktreeId diff --git a/src/main/codex-usage/scanner-attribution-memo.test.ts b/src/main/codex-usage/scanner-attribution-memo.test.ts new file mode 100644 index 00000000000..18a8326a981 --- /dev/null +++ b/src/main/codex-usage/scanner-attribution-memo.test.ts @@ -0,0 +1,163 @@ +import { mkdirSync, mkdtempSync, realpathSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import type * as NodeOs from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import type * as WorktreeLogic from '../ipc/worktree-logic' + +const { getPathMock, homedirMock, worktreePathComparisons } = vi.hoisted(() => ({ + getPathMock: vi.fn<(name: string) => string>(), + homedirMock: vi.fn<() => string>(), + worktreePathComparisons: { count: 0 } +})) + +vi.mock('electron', () => ({ + app: { + getPath: getPathMock + } +})) + +vi.mock('node:os', async () => { + const actual = await vi.importActual('node:os') + return { + ...actual, + homedir: homedirMock + } +}) + +vi.mock('../ipc/worktree-logic', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + areWorktreePathsEqual: (left: string, right: string) => { + worktreePathComparisons.count += 1 + return actual.areWorktreePathsEqual(left, right) + } + } +}) + +import { scanCodexUsageFiles } from './scanner' + +const WORKTREE_COUNT = 4 +const EVENTS_PER_FILE = 4 + +let fakeHomeDir: string +let userDataDir: string +let previousUserDataPath: string | undefined +const originalCodexHome = process.env.CODEX_HOME + +function usageRecord(timestamp: string, totalInputTokens: number): string { + return `${JSON.stringify({ + timestamp, + type: 'event_msg', + payload: { + type: 'token_count', + info: { + model: 'gpt-5-codex', + last_token_usage: { + input_tokens: 1, + cached_input_tokens: 0, + output_tokens: 0, + reasoning_output_tokens: 0, + total_tokens: 1 + }, + total_token_usage: { + input_tokens: totalInputTokens, + cached_input_tokens: 0, + output_tokens: 0, + reasoning_output_tokens: 0, + total_tokens: totalInputTokens + } + } + } + })}\n` +} + +function writeSessionFile( + sessionsDir: string, + sessionId: string, + cwd: string, + tokenOffset: number +): void { + // Why: event keys are content-derived, so identical records across files would be + // deduped by cross-file ownership and the scan would see one session, not three. + const records = [ + `${JSON.stringify({ type: 'session_meta', payload: { id: sessionId, cwd } })}\n`, + ...Array.from({ length: EVENTS_PER_FILE }, (_, index) => + usageRecord( + `2026-07-21T12:${String(index).padStart(2, '0')}:00.000Z`, + tokenOffset + index + 1 + ) + ) + ] + writeFileSync(join(sessionsDir, `${sessionId}.jsonl`), records.join(''), 'utf-8') +} + +beforeEach(() => { + delete process.env.CODEX_HOME + worktreePathComparisons.count = 0 + // Why: worktree canonicalization realpaths, so /var vs /private/var would never match. + fakeHomeDir = realpathSync(mkdtempSync(join(tmpdir(), 'orca-codex-memo-home-'))) + userDataDir = mkdtempSync(join(tmpdir(), 'orca-codex-memo-user-data-')) + previousUserDataPath = process.env.ORCA_USER_DATA_PATH + process.env.ORCA_USER_DATA_PATH = userDataDir + homedirMock.mockReturnValue(fakeHomeDir) + getPathMock.mockImplementation((name: string) => { + if (name === 'userData') { + return userDataDir + } + throw new Error(`unexpected app.getPath(${name})`) + }) +}) + +afterEach(() => { + rmSync(fakeHomeDir, { recursive: true, force: true }) + rmSync(userDataDir, { recursive: true, force: true }) + if (originalCodexHome === undefined) { + delete process.env.CODEX_HOME + } else { + process.env.CODEX_HOME = originalCodexHome + } + if (previousUserDataPath === undefined) { + delete process.env.ORCA_USER_DATA_PATH + } else { + process.env.ORCA_USER_DATA_PATH = previousUserDataPath + } + vi.clearAllMocks() +}) + +it('resolves each distinct cwd once per scan, not once per event', async () => { + const sessionsDir = join(fakeHomeDir, '.codex', 'sessions') + mkdirSync(sessionsDir, { recursive: true }) + const matchedCwd = join(fakeHomeDir, 'worktrees', 'repo-003', 'packages', 'app') + const unmatchedCwd = join(fakeHomeDir, 'elsewhere', 'project') + // Two files share a cwd so the memo must survive across files, not just within one. + writeSessionFile(sessionsDir, 'session-a', matchedCwd, 0) + writeSessionFile(sessionsDir, 'session-b', matchedCwd, 1_000) + writeSessionFile(sessionsDir, 'session-c', unmatchedCwd, 2_000) + const worktrees = Array.from({ length: WORKTREE_COUNT }, (_, index) => { + const worktreePath = join(fakeHomeDir, 'worktrees', `repo-${String(index).padStart(3, '0')}`) + mkdirSync(worktreePath, { recursive: true }) + return { + repoId: `repo-${index}`, + worktreeId: `repo-${index}::${worktreePath}`, + path: worktreePath, + displayName: `Repo ${index}` + } + }) + + const result = await scanCodexUsageFiles(worktrees, []) + + expect(result.sessions).toHaveLength(3) + const attributedWorktreeIds = new Set( + result.sessions.flatMap((session) => + session.locationBreakdown.map((location) => location.worktreeId) + ) + ) + expect(attributedWorktreeIds).toEqual( + new Set([`repo-3::${join(fakeHomeDir, 'worktrees', 'repo-003')}`, null]) + ) + // Two distinct cwds against every worktree; an unmemoized scan would pay this per event. + expect(worktreePathComparisons.count).toBeLessThanOrEqual(2 * WORKTREE_COUNT) + expect(worktreePathComparisons.count).toBeLessThan(EVENTS_PER_FILE * WORKTREE_COUNT) +}) diff --git a/src/main/codex-usage/scanner.test.ts b/src/main/codex-usage/scanner.test.ts index 746be0ffa15..296ed7d5229 100644 --- a/src/main/codex-usage/scanner.test.ts +++ b/src/main/codex-usage/scanner.test.ts @@ -10,6 +10,7 @@ vi.mock('electron', () => ({ } })) +import { createUsageWorktreeResolver } from '../usage/usage-worktree-resolver' import { attributeCodexUsageEvent } from './codex-usage-event-attribution' import { parseCodexUsageRecord } from './codex-usage-record-parser' @@ -208,22 +209,20 @@ describe('attributeCodexUsageEvent', () => { reasoningOutputTokens: 10, totalTokens: 125 }, - [ + await createUsageWorktreeResolver([ { repoId: 'repo-1', worktreeId: 'repo-1::/workspace/repo/app', path: '/workspace/repo/app', - displayName: 'App', - canonicalPath: '/workspace/repo/app' + displayName: 'App' }, { repoId: 'repo-2', worktreeId: 'repo-2::/workspace/repo/app2', path: '/workspace/repo/app2', - displayName: 'App 2', - canonicalPath: '/workspace/repo/app2' + displayName: 'App 2' } - ] + ]) ) expect(attributed?.projectKey).toBe('worktree:repo-2::/workspace/repo/app2') @@ -246,15 +245,14 @@ describe('attributeCodexUsageEvent', () => { reasoningOutputTokens: 10, totalTokens: 125 }, - [ + await createUsageWorktreeResolver([ { repoId: 'repo-1', worktreeId: 'repo-1::/workspace/repo', path: '/workspace/repo', - displayName: 'Repo', - canonicalPath: '/workspace/repo' + displayName: 'Repo' } - ] + ]) ) expect(attributed?.projectKey).toBe('worktree:repo-1::/workspace/repo') @@ -277,15 +275,14 @@ describe('attributeCodexUsageEvent', () => { reasoningOutputTokens: 10, totalTokens: 125 }, - [ + await createUsageWorktreeResolver([ { repoId: 'repo-1', worktreeId: 'repo-1::/workspace/repo', path: '/workspace/repo', - displayName: 'Repo', - canonicalPath: '/workspace/repo' + displayName: 'Repo' } - ] + ]) ) expect(attributed?.projectKey).toBe('cwd:/workspace/repo/../other/session') @@ -307,15 +304,14 @@ describe('attributeCodexUsageEvent', () => { reasoningOutputTokens: 10, totalTokens: 125 }, - [ + await createUsageWorktreeResolver([ { repoId: 'repo-1', worktreeId: 'repo-1::C:\\repo', path: 'C:\\repo', - displayName: 'Repo', - canonicalPath: 'C:\\repo' + displayName: 'Repo' } - ] + ]) ) expect(attributed?.projectKey).toBe('cwd:d:/other/repo') diff --git a/src/main/codex-usage/scanner.ts b/src/main/codex-usage/scanner.ts index d06a85f8901..0d5c1eba526 100644 --- a/src/main/codex-usage/scanner.ts +++ b/src/main/codex-usage/scanner.ts @@ -2,10 +2,12 @@ import { basename } from 'node:path' import { createReadStream } from 'node:fs' import { stat } from 'node:fs/promises' import { createInterface } from 'node:readline' -import { canonicalizeUsageWorktreePaths } from '../usage-worktree-canonicalizer' import { createUsageEventAggregation } from '../usage/usage-event-aggregation' import { - canonicalizePath, + createUsageWorktreeResolver, + type UsageWorktreeResolver +} from '../usage/usage-worktree-resolver' +import { getLegacySourceSkipBytesByPath, listCodexSessionFiles, yieldToEventLoop @@ -34,12 +36,6 @@ export async function getProcessedFileInfo(filePath: string): Promise { - return canonicalizeUsageWorktreePaths(worktrees, canonicalizePath) -} - type CodexUsageMetric = { hasInferredPricing: boolean } const codexUsageAggregation = createUsageEventAggregation< @@ -66,7 +62,7 @@ const { finalizeSessions, mergeSessions, mergeDailyAggregates, sortDailyAggregat export async function parseCodexUsageFile( filePath: string, - worktrees: (CodexUsageWorktreeRef & { canonicalPath: string })[], + resolveWorktree: UsageWorktreeResolver, options: { skipInitialBytes?: number; claimEventKey?: (eventKey: string) => boolean } = {} ): Promise { const processedFile = await getProcessedFileInfo(filePath) @@ -104,7 +100,7 @@ export async function parseCodexUsageFile( continue } ownedEventKeys.add(parsed.eventKey) - const attributed = await attributeCodexUsageEvent(parsed, worktrees) + const attributed = await attributeCodexUsageEvent(parsed, resolveWorktree) if (attributed) { events.push(attributed) } @@ -128,7 +124,8 @@ export async function scanCodexUsageFiles( }> { const files = await listCodexSessionFiles() const previousByPath = new Map(previousProcessedFiles.map((file) => [file.path, file])) - const worktreesWithCanonicalPaths = await buildWorktreesWithCanonicalPaths(worktrees) + // Why: one resolver for the whole scan so every file shares the per-cwd memo. + const resolveWorktree = await createUsageWorktreeResolver(worktrees) const legacySourceSkipBytesByPath = getLegacySourceSkipBytesByPath(files) const currentPaths = new Set(files) @@ -187,7 +184,7 @@ export async function scanCodexUsageFiles( const parsedByPath = new Map() for (const [index, filePath] of pathsToParse.entries()) { - const processed = await parseCodexUsageFile(filePath, worktreesWithCanonicalPaths, { + const processed = await parseCodexUsageFile(filePath, resolveWorktree, { skipInitialBytes: legacySourceSkipBytesByPath.get(filePath) ?? 0, claimEventKey: (eventKey) => { const owner = eventOwnerByKey.get(eventKey) diff --git a/src/main/opencode-usage/opencode-usage-worktree-attribution.ts b/src/main/opencode-usage/opencode-usage-worktree-attribution.ts index 03d0e563059..3f236ccc1d2 100644 --- a/src/main/opencode-usage/opencode-usage-worktree-attribution.ts +++ b/src/main/opencode-usage/opencode-usage-worktree-attribution.ts @@ -1,13 +1,6 @@ -import { realpath } from 'node:fs/promises' -import { posix, win32 } from 'node:path' -import { areWorktreePathsEqual } from '../ipc/worktree-logic' -import { canonicalizeUsageWorktreePaths } from '../usage-worktree-canonicalizer' -import { - looksLikeWindowsPath, - normalizeComparablePath, - normalizeFsPath -} from '../usage/usage-path-comparison' +import { normalizeComparablePath } from '../usage/usage-path-comparison' import type { UsageScanWorktreeRef } from '../usage/usage-provider-contract' +import type { UsageWorktreeResolver } from '../usage/usage-worktree-resolver' import type { OpenCodeUsageAttributedEvent, OpenCodeUsageParsedEvent } from './types' export type OpenCodeUsageWorktreeRef = UsageScanWorktreeRef @@ -34,60 +27,9 @@ function localDayFromTimestamp(timestamp: string): string | null { return `${year}-${month}-${day}` } -function isContainingPath(candidatePath: string, targetPath: string): boolean { - const useWin32 = looksLikeWindowsPath(candidatePath) || looksLikeWindowsPath(targetPath) - const relativePath = useWin32 - ? win32.relative(candidatePath, targetPath) - : posix.relative(candidatePath, targetPath) - if (!relativePath) { - return true - } - const isAbsoluteRelative = useWin32 - ? win32.isAbsolute(relativePath) - : posix.isAbsolute(relativePath) - const parentPrefix = useWin32 ? `..${win32.sep}` : `..${posix.sep}` - // Why: `..name` is a valid child path; only `..` and `../...` escape. - return ( - !isAbsoluteRelative && - relativePath !== '..' && - !relativePath.startsWith(parentPrefix) && - relativePath !== '.' - ) -} - -export async function buildWorktreesWithCanonicalPaths( - worktrees: OpenCodeUsageWorktreeRef[] -): Promise<(OpenCodeUsageWorktreeRef & { canonicalPath: string })[]> { - return canonicalizeUsageWorktreePaths(worktrees, canonicalizePath) -} - -async function canonicalizePath(pathValue: string): Promise { - try { - return normalizeFsPath(await realpath(pathValue)) - } catch { - return normalizeFsPath(pathValue) - } -} - -function findContainingWorktree( - cwd: string, - worktrees: (OpenCodeUsageWorktreeRef & { canonicalPath: string })[] -): OpenCodeUsageWorktreeRef | null { - const normalizedCwd = normalizeFsPath(cwd) - for (const worktree of worktrees) { - if (areWorktreePathsEqual(worktree.canonicalPath, normalizedCwd)) { - return worktree - } - if (isContainingPath(worktree.canonicalPath, normalizedCwd)) { - return worktree - } - } - return null -} - export async function attributeOpenCodeUsageEvent( event: OpenCodeUsageParsedEvent, - worktrees: (OpenCodeUsageWorktreeRef & { canonicalPath: string })[] + resolveWorktree: UsageWorktreeResolver ): Promise { const day = localDayFromTimestamp(event.timestamp) if (!day) { @@ -100,7 +42,7 @@ export async function attributeOpenCodeUsageEvent( let projectLabel = getDefaultProjectLabel(event.cwd) if (event.cwd) { - const worktree = findContainingWorktree(event.cwd, worktrees) + const worktree = resolveWorktree(event.cwd) if (worktree) { repoId = worktree.repoId worktreeId = worktree.worktreeId diff --git a/src/main/opencode-usage/scanner.test.ts b/src/main/opencode-usage/scanner.test.ts index f4ce66b51e4..9ee01327964 100644 --- a/src/main/opencode-usage/scanner.test.ts +++ b/src/main/opencode-usage/scanner.test.ts @@ -5,6 +5,7 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest' import Database from '../sqlite/sync-database' import { listOpenCodeDatabases } from './opencode-database-discovery' import { parseOpenCodeUsageRow } from './opencode-usage-row-parsing' +import { createUsageWorktreeResolver } from '../usage/usage-worktree-resolver' import { attributeOpenCodeUsageEvent } from './opencode-usage-worktree-attribution' import { parseOpenCodeUsageDatabase, scanOpenCodeUsageDatabases } from './scanner' @@ -19,16 +20,15 @@ function createTempDb(): { db: Database.Database; path: string } { return { db: new Database(path), path } } -function worktrees() { - return [ +async function resolveWorktree() { + return createUsageWorktreeResolver([ { repoId: 'repo-1', worktreeId: 'repo-1::/workspace/repo', path: WORKTREE, - displayName: 'Repo', - canonicalPath: WORKTREE + displayName: 'Repo' } - ] + ]) } function createSessionTotalsSchema(db: Database.Database): void { @@ -138,7 +138,7 @@ describe('attributeOpenCodeUsageEvent', () => { it('attributes cwd paths under dotdot-prefixed child directories to the worktree', async () => { const attributed = await attributeOpenCodeUsageEvent( usageEvent(`${WORKTREE}/..fixtures/session`), - worktrees() + await resolveWorktree() ) expect(attributed?.projectKey).toBe('worktree:repo-1::/workspace/repo') @@ -149,7 +149,7 @@ describe('attributeOpenCodeUsageEvent', () => { it('does not attribute true parent-directory escapes to the worktree', async () => { const attributed = await attributeOpenCodeUsageEvent( usageEvent(`${WORKTREE}/../other/session`), - worktrees() + await resolveWorktree() ) expect(attributed?.projectKey).toBe('cwd:/workspace/repo/../other/session') @@ -157,15 +157,17 @@ describe('attributeOpenCodeUsageEvent', () => { }) it('does not treat different Windows drives as containing paths', async () => { - const attributed = await attributeOpenCodeUsageEvent(usageEvent('D:\\other\\repo'), [ - { - repoId: 'repo-1', - worktreeId: 'repo-1::C:\\repo', - path: 'C:\\repo', - displayName: 'Repo', - canonicalPath: 'C:\\repo' - } - ]) + const attributed = await attributeOpenCodeUsageEvent( + usageEvent('D:\\other\\repo'), + await createUsageWorktreeResolver([ + { + repoId: 'repo-1', + worktreeId: 'repo-1::C:\\repo', + path: 'C:\\repo', + displayName: 'Repo' + } + ]) + ) expect(attributed?.projectKey).toBe('cwd:d:/other/repo') expect(attributed?.worktreeId).toBeNull() @@ -222,7 +224,7 @@ describe('parseOpenCodeUsageDatabase', () => { ) db.close() - const parsed = await parseOpenCodeUsageDatabase(path, worktrees()) + const parsed = await parseOpenCodeUsageDatabase(path, await resolveWorktree()) expect(parsed.sessions).toHaveLength(1) expect(parsed.sessions[0]).toMatchObject({ @@ -292,7 +294,7 @@ describe('parseOpenCodeUsageDatabase', () => { ) db.close() - const parsed = await parseOpenCodeUsageDatabase(path, worktrees()) + const parsed = await parseOpenCodeUsageDatabase(path, await resolveWorktree()) expect(parsed.sessions[0]).toMatchObject({ primaryModel: 'openai/gpt-5.5', @@ -308,7 +310,7 @@ describe('parseOpenCodeUsageDatabase', () => { insertSessionTotalsRow(db, 'session-1', 1000) db.close() - const parsed = await parseOpenCodeUsageDatabase(path, worktrees()) + const parsed = await parseOpenCodeUsageDatabase(path, await resolveWorktree()) expect(parsed.ownedSessionIds).toEqual(['session-1']) }) @@ -372,7 +374,7 @@ describe('parseOpenCodeUsageDatabase', () => { ) db.close() - const parsed = await parseOpenCodeUsageDatabase(path, worktrees()) + const parsed = await parseOpenCodeUsageDatabase(path, await resolveWorktree()) expect(parsed.sessions[0]?.totalTokens).toBe(120) expect(parsed.sessions[0]?.eventCount).toBe(1) diff --git a/src/main/opencode-usage/scanner.ts b/src/main/opencode-usage/scanner.ts index bd1ae8e3670..369fb407a00 100644 --- a/src/main/opencode-usage/scanner.ts +++ b/src/main/opencode-usage/scanner.ts @@ -1,6 +1,10 @@ import { yieldToEventLoop } from '../../shared/event-loop-yield' import Database from '../sqlite/sync-database' import { createUsageEventAggregation } from '../usage/usage-event-aggregation' +import { + createUsageWorktreeResolver, + type UsageWorktreeResolver +} from '../usage/usage-worktree-resolver' import { compareOpenCodeClaimPriority, getProcessedDatabaseInfo, @@ -10,7 +14,6 @@ import { parseOpenCodeUsageRow } from './opencode-usage-row-parsing' import { selectUsageRows } from './opencode-usage-row-queries' import { attributeOpenCodeUsageEvent, - buildWorktreesWithCanonicalPaths, type OpenCodeUsageWorktreeRef } from './opencode-usage-worktree-attribution' import type { @@ -50,7 +53,7 @@ const { finalizeSessions, mergeSessions, mergeDailyAggregates, sortDailyAggregat export async function parseOpenCodeUsageDatabase( dbPath: string, - worktrees: (OpenCodeUsageWorktreeRef & { canonicalPath: string })[], + resolveWorktree: UsageWorktreeResolver, options: { claimSession?: (sessionId: string) => boolean } = {} ): Promise { const processedDatabase = await getProcessedDatabaseInfo(dbPath) @@ -76,7 +79,7 @@ export async function parseOpenCodeUsageDatabase( hasDeferredClaims = true continue } - const attributed = await attributeOpenCodeUsageEvent(parsed, worktrees) + const attributed = await attributeOpenCodeUsageEvent(parsed, resolveWorktree) if (attributed) { events.push(attributed) } @@ -106,7 +109,8 @@ export async function scanOpenCodeUsageDatabases( const previousByPath = new Map( previousProcessedDatabases.map((database) => [database.path, database]) ) - const worktreesWithCanonicalPaths = await buildWorktreesWithCanonicalPaths(worktrees) + // Why: one resolver for the whole scan so every database shares the per-cwd memo. + const resolveWorktree = await createUsageWorktreeResolver(worktrees) const currentPaths = new Set(dbPaths) // Why: when a database that owned sessions is deleted, remaining siblings @@ -182,7 +186,7 @@ export async function scanOpenCodeUsageDatabases( const parsedByPath = new Map() const orderedPathsToParse = [...pathsToParse].sort(compareOpenCodeClaimPriority) for (const [index, dbPath] of orderedPathsToParse.entries()) { - const processed = await parseOpenCodeUsageDatabase(dbPath, worktreesWithCanonicalPaths, { + const processed = await parseOpenCodeUsageDatabase(dbPath, resolveWorktree, { claimSession: (sessionId) => { const owner = sessionOwnerById.get(sessionId) if (owner !== undefined && owner !== dbPath) { diff --git a/src/main/usage/usage-worktree-resolver.test.ts b/src/main/usage/usage-worktree-resolver.test.ts new file mode 100644 index 00000000000..f53cb74c2af --- /dev/null +++ b/src/main/usage/usage-worktree-resolver.test.ts @@ -0,0 +1,87 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as WorktreeLogic from '../ipc/worktree-logic' +import type { UsageScanWorktreeRef } from './usage-provider-contract' + +const { worktreePathComparisons } = vi.hoisted(() => ({ + worktreePathComparisons: { count: 0 } +})) + +vi.mock('../ipc/worktree-logic', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + areWorktreePathsEqual: (left: string, right: string) => { + worktreePathComparisons.count += 1 + return actual.areWorktreePathsEqual(left, right) + } + } +}) + +import { createUsageWorktreeResolver } from './usage-worktree-resolver' + +function worktree(path: string, index: number): UsageScanWorktreeRef { + return { + repoId: `repo-${index}`, + worktreeId: `repo-${index}::${path}`, + path, + displayName: `Repo ${index}` + } +} + +describe('createUsageWorktreeResolver', () => { + beforeEach(() => { + worktreePathComparisons.count = 0 + }) + + it('walks the worktree list once per distinct cwd, including misses', async () => { + const resolveWorktree = await createUsageWorktreeResolver( + Array.from({ length: 50 }, (_, index) => + worktree(`/repo-${String(index).padStart(3, '0')}`, index) + ) + ) + const attribute = (event: number): string | null => { + const cwd = event % 2 === 0 ? '/repo-049/nested/pkg' : '/outside/project' + return resolveWorktree(cwd)?.worktreeId ?? null + } + + expect(attribute(0)).toBe('repo-49::/repo-049') + expect(attribute(1)).toBeNull() + const afterFirstOfEachCwd = worktreePathComparisons.count + expect(afterFirstOfEachCwd).toBeGreaterThan(0) + expect(afterFirstOfEachCwd).toBeLessThanOrEqual(100) + + for (let event = 2; event < 1_000; event++) { + expect(attribute(event)).toBe(event % 2 === 0 ? 'repo-49::/repo-049' : null) + } + + // Two distinct cwds walked the list once each; 998 more events cost nothing. + expect(worktreePathComparisons.count).toBe(afterFirstOfEachCwd) + }) + + it('keeps containment semantics unchanged', async () => { + const resolveWorktree = await createUsageWorktreeResolver([worktree('/workspace/repo', 1)]) + + expect(resolveWorktree('/workspace/repo')?.worktreeId).toBe('repo-1::/workspace/repo') + expect(resolveWorktree('/workspace/repo/packages/app')?.worktreeId).toBe( + 'repo-1::/workspace/repo' + ) + // `..name` is a child directory; `..` escapes. + expect(resolveWorktree('/workspace/repo/..fixtures/session')?.worktreeId).toBe( + 'repo-1::/workspace/repo' + ) + expect(resolveWorktree('/workspace/repo/../other/session')).toBeNull() + expect(resolveWorktree('/workspace/repo-sibling')).toBeNull() + }) + + it('does not treat a different Windows drive as contained', async () => { + const resolveWorktree = await createUsageWorktreeResolver([worktree('C:\\repo', 1)]) + + expect(resolveWorktree('C:\\repo\\packages\\app')?.worktreeId).toBe('repo-1::C:\\repo') + expect(resolveWorktree('D:\\other\\repo')).toBeNull() + }) + + it('resolves nothing when no worktree is known', async () => { + const resolveWorktree = await createUsageWorktreeResolver([]) + expect(resolveWorktree('/workspace/repo')).toBeNull() + }) +}) diff --git a/src/main/usage/usage-worktree-resolver.ts b/src/main/usage/usage-worktree-resolver.ts new file mode 100644 index 00000000000..1ae49f2a889 --- /dev/null +++ b/src/main/usage/usage-worktree-resolver.ts @@ -0,0 +1,84 @@ +import { realpath } from 'node:fs/promises' +import { posix, win32 } from 'node:path' +import { areWorktreePathsEqual } from '../ipc/worktree-logic' +import { canonicalizeUsageWorktreePaths } from '../usage-worktree-canonicalizer' +import { looksLikeWindowsPath, normalizeFsPath } from './usage-path-comparison' +import type { UsageScanWorktreeRef } from './usage-provider-contract' + +type CanonicalizedUsageWorktreeRef = UsageScanWorktreeRef & { canonicalPath: string } + +/** Maps an event's `cwd` to the worktree that contains it, or null when it is outside every one. */ +export type UsageWorktreeResolver = (cwd: string) => UsageScanWorktreeRef | null + +function isContainingPath(candidatePath: string, targetPath: string): boolean { + const useWin32 = looksLikeWindowsPath(candidatePath) || looksLikeWindowsPath(targetPath) + const relativePath = useWin32 + ? win32.relative(candidatePath, targetPath) + : posix.relative(candidatePath, targetPath) + if (!relativePath) { + return true + } + // Why: on Windows, `path.relative('C:\\repo', 'D:\\other')` returns an + // absolute `D:\\other` path instead of a `..`-prefixed relative. Treating + // that as "contained" would attribute off-drive usage to the wrong + // Orca worktree. + const isAbsoluteRelative = useWin32 + ? win32.isAbsolute(relativePath) + : posix.isAbsolute(relativePath) + const parentPrefix = useWin32 ? `..${win32.sep}` : `..${posix.sep}` + // Why: `..name` is a valid child path; only `..` and `../...` escape. + return ( + !isAbsoluteRelative && + relativePath !== '..' && + !relativePath.startsWith(parentPrefix) && + relativePath !== '.' + ) +} + +async function canonicalizePath(pathValue: string): Promise { + try { + return normalizeFsPath(await realpath(pathValue)) + } catch { + return normalizeFsPath(pathValue) + } +} + +function findContainingWorktree( + cwd: string, + worktrees: readonly CanonicalizedUsageWorktreeRef[] +): UsageScanWorktreeRef | null { + const normalizedCwd = normalizeFsPath(cwd) + for (const worktree of worktrees) { + if (areWorktreePathsEqual(worktree.canonicalPath, normalizedCwd)) { + return worktree + } + if (isContainingPath(worktree.canonicalPath, normalizedCwd)) { + return worktree + } + } + return null +} + +/** + * Resolver for one scan, memoized per `cwd`. + * + * Why: attribution runs per event but a corpus holds only a few hundred distinct cwds, so an + * unmemoized search costs `events × worktrees` — 1.6M events against a few hundred remembered + * worktrees is minutes of main-thread CPU (STA-7724). + */ +export async function createUsageWorktreeResolver( + worktrees: readonly UsageScanWorktreeRef[] +): Promise { + const canonicalized = await canonicalizeUsageWorktreePaths(worktrees, canonicalizePath) + const worktreeByCwd = new Map() + return (cwd) => { + const memoized = worktreeByCwd.get(cwd) + // Why: a cwd outside every worktree memoizes as null, so only `undefined` is a miss. + if (memoized !== undefined) { + return memoized + } + const resolved = findContainingWorktree(cwd, canonicalized) + worktreeByCwd.set(cwd, resolved) + return resolved + } +}