From 5a1acfec1722cf3725dc8e5b096dead2c5302284 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Fri, 4 Sep 2026 20:03:51 -0700 Subject: [PATCH 01/26] fix(native-chat): make document paths and links clickable in chat (#18712) * fix(chat): make assistant file paths clickable * fix(chat): tighten native file link handling * fix(chat): link prose-joined relative paths separately * fix(chat): preserve complete Unicode file links * fix(chat): preserve links before sentence punctuation * test(chat): align structured session link props * fix(native-chat): harden generated file links * fix(native-chat): reject reference-number false positives * test(native-chat): align structured session parity * perf(native-chat): bound file-link detection --------- Co-authored-by: Merge Sim --- .../native-chat/NativeChatMessageList.tsx | 1 + .../NativeChatStructuredSession.test.tsx | 19 +- .../NativeChatStructuredSession.tsx | 2 +- ...turedAgentSessionPaneOverlayLayer.test.tsx | 6 +- ...StructuredAgentSessionPaneOverlayLayer.tsx | 14 +- .../native-chat/native-chat-view-types.ts | 1 - .../CommentMarkdown.link-click.test.tsx | 454 ++++++++++++++++++ .../components/sidebar/CommentMarkdown.tsx | 13 +- .../comment-markdown-element-renderers.tsx | 21 +- ...comment-markdown-native-chat-file-links.ts | 243 ++++++++++ .../TerminalPaneNativeChatPortal.tsx | 1 - src/renderer/src/lib/terminal-links.test.ts | 18 + src/renderer/src/lib/terminal-links.ts | 20 +- src/shared/native-chat-href-routing.test.ts | 31 +- src/shared/native-chat-href-routing.ts | 35 +- 15 files changed, 824 insertions(+), 55 deletions(-) create mode 100644 src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx index 8336f340b8c..53ed2cb8244 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx @@ -156,6 +156,7 @@ function MessageRow({ className="text-sm" onLinkClick={onLinkClick} allowFileUriLinks={allowFileUriLinks} + linkifyFilePaths={onLinkClick !== undefined} /> ) : null} {tools.length > 0 ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx index 70afd6758aa..356a482dfa3 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx @@ -165,7 +165,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-paste" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -176,15 +175,14 @@ describe('NativeChatStructuredSession', () => { expect(mocks.pasteFromClipboard).toHaveBeenCalledOnce() }) - it('wires local structured file links through the native chat opener', () => { + it('wires remote structured file links through the host-aware native chat opener', () => { render( ) @@ -204,7 +202,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-parity" target={{ kind: 'local' }} agent={agent} - allowFileUriLinks /> ) @@ -221,7 +218,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-1" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) const dispatchCommand = mocks.composerProps?.structuredTransport?.dispatchCommand as @@ -257,7 +253,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-1" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -289,7 +284,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-wedge" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -321,7 +315,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-probe-flag" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -354,7 +347,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-parked" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -404,7 +396,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-churn" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) const { rerender } = render(makeView()) @@ -458,7 +449,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-target-switch" target={target} agent="codex" - allowFileUriLinks /> ) const { rerender } = render(makeView({ kind: 'local' })) @@ -493,7 +483,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-forced" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -532,7 +521,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-pending" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -562,7 +550,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-budget" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -633,7 +620,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-questions" target={{ kind: 'local' }} agent="claude" - allowFileUriLinks={false} /> ) @@ -692,7 +678,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-legacy-question" target={{ kind: 'local' }} agent="claude" - allowFileUriLinks={false} /> ) diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index 34ee980f0b3..4c5038b6dd6 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -80,7 +80,7 @@ export function NativeChatStructuredSession( const fontScale = useNativeChatFontScale(viewState.kind === 'ready') const fileLinkContext = useNativeChatFileLinkContext(props.tabId) const imageRuntimeContext = useNativeChatImageRuntimeContext(props.tabId) - const fileLinkClick = useNativeChatFileLinkClick(props.allowFileUriLinks ? fileLinkContext : null) + const fileLinkClick = useNativeChatFileLinkClick(fileLinkContext) const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null const questions = diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx index 3cf25030aa1..05101f14fd1 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx @@ -8,7 +8,6 @@ type MockAppState = { unifiedTabsByWorktree: Record groupsByWorktree: Record runtimeEnvironmentId: string | null - executionHostId: string focusGroup: (worktreeId: string, groupId: string) => void } @@ -26,7 +25,6 @@ vi.mock('@/store', async () => { unifiedTabsByWorktree: {}, groupsByWorktree: {}, runtimeEnvironmentId: null, - executionHostId: 'local', focusGroup: mocks.focusGroup })) mocks.store = useAppStore @@ -34,8 +32,7 @@ vi.mock('@/store', async () => { }) vi.mock('@/lib/worktree-runtime-owner', () => ({ - getRuntimeEnvironmentIdForWorktree: (state: MockAppState) => state.runtimeEnvironmentId, - getExecutionHostIdForWorktree: (state: MockAppState) => state.executionHostId + getRuntimeEnvironmentIdForWorktree: (state: MockAppState) => state.runtimeEnvironmentId })) vi.mock('@/runtime/runtime-rpc-client', () => ({ @@ -177,7 +174,6 @@ function createState(activeTabId: string): MockAppState { }, groupsByWorktree: { [WORKTREE_ID]: [createGroup(activeTabId)] }, runtimeEnvironmentId: null, - executionHostId: 'local', focusGroup: mocks.focusGroup } } diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx index cf6a0e0b4d1..01b8c256277 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx @@ -3,10 +3,7 @@ import { useShallow } from 'zustand/react/shallow' import type { Tab, TabGroup } from '../../../../shared/tab-types' import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import { useAppStore } from '@/store' -import { - getExecutionHostIdForWorktree, - getRuntimeEnvironmentIdForWorktree -} from '@/lib/worktree-runtime-owner' +import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import { getActiveRuntimeTarget, type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { tabGroupBodyAnchorName } from '../tab-group/tab-group-body-anchor' import NativeChatView from './NativeChatView' @@ -24,14 +21,12 @@ const StructuredAgentSessionOverlaySlot = memo(function StructuredAgentSessionOv groupId, isActive, target, - allowFileUriLinks, onFocusOwningGroup }: { tab: StructuredAgentSessionTab groupId: string | undefined isActive: boolean target: RuntimeClientTarget - allowFileUriLinks: boolean onFocusOwningGroup: ((groupId: string) => void) | undefined }): React.JSX.Element { const anchorName = groupId !== undefined ? tabGroupBodyAnchorName(groupId) : undefined @@ -74,7 +69,6 @@ const StructuredAgentSessionOverlaySlot = memo(function StructuredAgentSessionOv agent={tab.agentSessionAgent} isVisible={isActive} target={target} - allowFileUriLinks={allowFileUriLinks} /> ) @@ -88,12 +82,11 @@ const StructuredAgentSessionPaneOverlayLayer = memo( worktreeId: string isWorktreeActive: boolean }): React.JSX.Element { - const { unifiedTabs, groups, runtimeEnvironmentId, allowFileUriLinks } = useAppStore( + const { unifiedTabs, groups, runtimeEnvironmentId } = useAppStore( useShallow((state) => ({ unifiedTabs: state.unifiedTabsByWorktree[worktreeId] ?? EMPTY_UNIFIED_TABS, groups: state.groupsByWorktree[worktreeId] ?? EMPTY_GROUPS, - runtimeEnvironmentId: getRuntimeEnvironmentIdForWorktree(state, worktreeId), - allowFileUriLinks: getExecutionHostIdForWorktree(state, worktreeId) === 'local' + runtimeEnvironmentId: getRuntimeEnvironmentIdForWorktree(state, worktreeId) })) ) const focusGroup = useAppStore((state) => state.focusGroup) @@ -128,7 +121,6 @@ const StructuredAgentSessionPaneOverlayLayer = memo( groupId={tab.groupId} isActive={Boolean(isWorktreeActive && groupActiveTabById.get(tab.groupId) === tab.id)} target={target} - allowFileUriLinks={allowFileUriLinks} onFocusOwningGroup={focusOwningGroup} /> ))} diff --git a/src/renderer/src/components/native-chat/native-chat-view-types.ts b/src/renderer/src/components/native-chat/native-chat-view-types.ts index de86ecbe228..920bede7028 100644 --- a/src/renderer/src/components/native-chat/native-chat-view-types.ts +++ b/src/renderer/src/components/native-chat/native-chat-view-types.ts @@ -42,7 +42,6 @@ export type NativeChatStructuredViewProps = NativeChatOrchestrationProps & { target: RuntimeClientTarget agent: AgentType isVisible: boolean - allowFileUriLinks: boolean contextMenuActions?: Omit } diff --git a/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx b/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx index 575d4343c3b..a3f7c7aad1f 100644 --- a/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx +++ b/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx @@ -3,6 +3,10 @@ import { act } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, describe, expect, it, vi } from 'vitest' +import { + NATIVE_CHAT_FILE_HREF_PREFIX, + routeNativeChatHref +} from '../../../../shared/native-chat-href-routing' import CommentMarkdown from './CommentMarkdown' describe('CommentMarkdown link click handler', () => { @@ -48,6 +52,72 @@ describe('CommentMarkdown link click handler', () => { expect(event.defaultPrevented).toBe(true) }) + it('intercepts auxiliary clicks on generated native file links', () => { + const onLinkClick = vi.fn((event: React.MouseEvent) => { + event.preventDefault() + }) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + const event = new window.MouseEvent('auxclick', { + bubbles: true, + cancelable: true, + button: 1 + }) + + act(() => { + anchor?.dispatchEvent(event) + }) + + expect(onLinkClick).toHaveBeenCalledWith(expect.any(Object), expect.stringMatching(/^#orca-/)) + expect(event.defaultPrevented).toBe(true) + }) + + it('does not activate generated native file links on right-click', () => { + const onLinkClick = vi.fn() + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + const event = new window.MouseEvent('auxclick', { + bubbles: true, + cancelable: true, + button: 2 + }) + + act(() => { + anchor?.dispatchEvent(event) + }) + + expect(onLinkClick).not.toHaveBeenCalled() + expect(event.defaultPrevented).toBe(false) + }) + it('sanitizes file URI links unless the caller opts in', () => { container = document.createElement('div') document.body.appendChild(container) @@ -146,4 +216,388 @@ describe('CommentMarkdown link click handler', () => { expect(onLinkClick).toHaveBeenCalledWith(expect.any(Object), 'assets/diagram.png') expect(event.defaultPrevented).toBe(true) }) + + it('linkifies bare POSIX and Windows document paths without an extension allowlist', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const routes = Array.from(container.querySelectorAll('a')).map((anchor) => + routeNativeChatHref(anchor.getAttribute('href')) + ) + expect(routes).toEqual([ + { kind: 'file', pathText: '/tmp/sta-6481-explainer.html', line: null }, + { kind: 'file', pathText: 'docs/review.docx', line: null }, + { kind: 'file', pathText: String.raw`C:\Reports\final.pages`, line: null }, + { kind: 'file', pathText: './scripts/release', line: null }, + { kind: 'file', pathText: 'src/release:12', line: null } + ]) + }) + + it('makes an inline-code file path clickable while preserving code styling', () => { + const onLinkClick = vi.fn((event: React.MouseEvent) => event.preventDefault()) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const code = container.querySelector('code') + const anchor = code?.closest('a') + expect(anchor).not.toBeNull() + expect(routeNativeChatHref(anchor?.getAttribute('href'))).toEqual({ + kind: 'file', + pathText: String.raw`C:\Reports\release.docx`, + line: null + }) + + act(() => { + anchor?.dispatchEvent(new window.MouseEvent('click', { bubbles: true, cancelable: true })) + }) + expect(onLinkClick).toHaveBeenCalledOnce() + }) + + it('leaves prose-shaped slash tokens and numeric versions unlinked', () => { + const proseFalsePositives = ['and/or', 'TCP/IP', '24/7', 'N/A', 'km/h', 'A/B test'] + const inlineCodeFalsePositives = ['origin/main', 'v1.2.3', '1.0'] + const quotedFalsePositives = ['"and/or"', '"A/B test"'] + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + `\`${value}\``).join(', ')}; ${quotedFalsePositives.join(', ')}`} + onLinkClick={vi.fn()} + linkifyFilePaths + /> + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + for (const value of proseFalsePositives) { + expect(container.textContent).toContain(value) + } + for (const value of quotedFalsePositives) { + expect(container.textContent).toContain(value) + } + expect(Array.from(container.querySelectorAll('code')).map((code) => code.textContent)).toEqual( + inlineCodeFalsePositives + ) + }) + + it('links each relative path separately when prose joins them', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['src/foo.ts', 'src/bar.ts', 'docs/My Folder/notes.md'] + ) + }) + + it('links quoted spaced-first-segment paths around apostrophes', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchors = container.querySelectorAll('a') + expect(Array.from(anchors).map((anchor) => anchor.textContent)).toEqual([ + "Brennan's Folder/notes.md", + 'My Folder/guide.md' + ]) + expect(container.textContent).toBe( + "Don't skip \"Brennan's Folder/notes.md\"; open 'My Folder/guide.md'." + ) + expect( + Array.from(anchors).map((anchor) => routeNativeChatHref(anchor.getAttribute('href'))) + ).toEqual([ + { kind: 'file', pathText: "Brennan's Folder/notes.md", line: null }, + { kind: 'file', pathText: 'My Folder/guide.md', line: null } + ]) + }) + + it('links a spaced-first-segment relative path when inline code disambiguates it', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + expect(anchor?.textContent).toBe('My Folder/notes.md') + expect(routeNativeChatHref(anchor?.getAttribute('href'))).toEqual({ + kind: 'file', + pathText: 'My Folder/notes.md', + line: null + }) + }) + + it('requires path shape before a spaced line suffix can make a link', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + expect(container.querySelector('code')?.textContent).toBe('aspect 16:9') + expect(container.textContent).toContain('"John 3:16"') + }) + + it('preserves line suffixes on valid spaced path shapes', () => { + const content = + 'Open "My Folder/notes:12", `My Notes.md:7`, and "C:\\My Folder\\notes.txt:12:3".' + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchors = Array.from(container.querySelectorAll('a')) + expect(anchors.map((anchor) => anchor.textContent)).toEqual([ + 'My Folder/notes:12', + 'My Notes.md:7', + String.raw`C:\My Folder\notes.txt:12:3` + ]) + expect(anchors.map((anchor) => routeNativeChatHref(anchor.getAttribute('href')))).toEqual([ + { kind: 'file', pathText: 'My Folder/notes:12', line: null }, + { kind: 'file', pathText: 'My Notes.md:7', line: null }, + { kind: 'file', pathText: String.raw`C:\My Folder\notes.txt:12:3`, line: null } + ]) + }) + + it('links complete Unicode paths and extensions that begin with a digit', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['/tmp/报告.html', 'docs/报告/file.html', 'docs/café/report.pdf', 'docs/archive.7z'] + ) + }) + + it('never links an ASCII suffix inside a path containing an unsupported character', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + expect(container.textContent).toContain('/tmp/$draft/report.html') + }) + + it('links paths before common sentence punctuation', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['src/foo.ts', 'docs/guide.md', 'assets/report.pdf'] + ) + }) + + it('links paths after CLI assignment and before Unicode sentence punctuation', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['./config.yaml', 'docs/指南.md', 'docs/报告.pdf'] + ) + }) + + it('does not link partial paths across unsupported punctuation', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + }) + + it('prevents the default action for an unresolved internal file href', () => { + const onLinkClick = vi.fn() + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + expect(anchor?.getAttribute('href')).toMatch(new RegExp(`^${NATIVE_CHAT_FILE_HREF_PREFIX}`)) + const event = new window.MouseEvent('click', { bubbles: true, cancelable: true }) + + act(() => { + anchor?.dispatchEvent(event) + }) + + expect(onLinkClick).toHaveBeenCalledOnce() + expect(event.defaultPrevented).toBe(true) + }) + + it('normalizes Windows markdown hrefs but leaves fenced paths as source text', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchors = container.querySelectorAll('a') + expect(anchors).toHaveLength(1) + expect(routeNativeChatHref(anchors[0]?.getAttribute('href'))).toEqual({ + kind: 'file', + pathText: String.raw`C:\Reports\summary.pdf`, + line: null + }) + expect(container.querySelector('pre')?.textContent).toContain('/tmp/not-a-link.html') + }) }) diff --git a/src/renderer/src/components/sidebar/CommentMarkdown.tsx b/src/renderer/src/components/sidebar/CommentMarkdown.tsx index 8673baedefa..0999c23c8c6 100644 --- a/src/renderer/src/components/sidebar/CommentMarkdown.tsx +++ b/src/renderer/src/components/sidebar/CommentMarkdown.tsx @@ -13,6 +13,7 @@ import { isTrustedCompactImageSrc, type CommentMarkdownLinkClickHandler } from './comment-markdown-element-renderers' +import { remarkNativeChatFileLinks } from './comment-markdown-native-chat-file-links' export type { CommentMarkdownLinkClickHandler } from './comment-markdown-element-renderers' @@ -185,6 +186,7 @@ type CommentMarkdownProps = React.ComponentPropsWithoutRef<'div'> & { githubRepo?: GitHubRepoReference | null onLinkClick?: CommentMarkdownLinkClickHandler allowFileUriLinks?: boolean + linkifyFilePaths?: boolean expandImages?: boolean } @@ -200,6 +202,7 @@ const CommentMarkdown = React.memo( githubRepo, onLinkClick, allowFileUriLinks = false, + linkifyFilePaths = false, expandImages = false, ...rest }, @@ -217,10 +220,12 @@ const CommentMarkdown = React.memo( ? createDocumentCommentMarkdownComponents(onLinkClick) : createCompactCommentMarkdownComponents(onLinkClick, expandImages) }, [expandImages, variant, onLinkClick]) - const activeRemarkPlugins = React.useMemo( - () => (githubRepo ? [...remarkPlugins, remarkGitHubReferences(githubRepo)] : remarkPlugins), - [githubRepo] - ) + const activeRemarkPlugins = React.useMemo(() => { + const plugins = linkifyFilePaths + ? [...remarkPlugins, remarkNativeChatFileLinks] + : remarkPlugins + return githubRepo ? [...plugins, remarkGitHubReferences(githubRepo)] : plugins + }, [githubRepo, linkifyFilePaths]) return (
, + href: string | undefined, + onLinkClick: CommentMarkdownLinkClickHandler | undefined +): void { + if (event.button === 1) { + handleMarkdownAnchorClick(event, href, onLinkClick) + } +} + function handleMarkdownImageClick( event: React.MouseEvent, src: string | undefined, @@ -65,6 +80,7 @@ export function createCompactCommentMarkdownComponents( rel="noreferrer" className="underline underline-offset-2 text-foreground/80 hover:text-foreground" onClick={(e) => handleMarkdownAnchorClick(e, href, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, href, onLinkClick)} > {children} @@ -154,6 +170,7 @@ export function createCompactCommentMarkdownComponents( rel="noreferrer" className="underline underline-offset-2 text-foreground/80 hover:text-foreground" onClick={(e) => handleMarkdownAnchorClick(e, src, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, src, onLinkClick)} > {alt || src} @@ -184,6 +201,7 @@ export function createCompactCommentMarkdownComponents( target="_blank" rel="noreferrer" onClick={(e) => handleMarkdownAnchorClick(e, src, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, src, onLinkClick)} > {image} @@ -221,6 +239,7 @@ export function createDocumentCommentMarkdownComponents( rel="noreferrer" className="break-all text-primary underline underline-offset-2 hover:text-primary/80" onClick={(e) => handleMarkdownAnchorClick(e, href, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, href, onLinkClick)} > {children} diff --git a/src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts b/src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts new file mode 100644 index 00000000000..ee81b859dcc --- /dev/null +++ b/src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts @@ -0,0 +1,243 @@ +import { + createNativeChatFileHref, + routeNativeChatHref +} from '../../../../shared/native-chat-href-routing' +import { parseFileLinkLocation } from '../../../../shared/file-link-location' +import { extractTerminalFileLinks, type ParsedTerminalFileLink } from '@/lib/terminal-links' + +type MarkdownNode = { + type: string + value?: string + url?: string + children?: MarkdownNode[] +} + +const ROOTED_PATH_PREFIX_PATTERN = /^(?:~[\\/]|\.{1,2}[\\/]|[\\/]|[A-Za-z]:[\\/])/ + +function isLinkifiableFile(link: ParsedTerminalFileLink, requireSeparator: boolean): boolean { + const hasRootedPrefix = ROOTED_PATH_PREFIX_PATTERN.test(link.pathText) + const hasLineSuffix = link.line !== null || link.column !== null + const hasAlphabeticExtension = /\.[\p{L}][\p{L}\p{N}\p{M}_+-]*$/u.test(link.pathText) + const hasPathExtension = /\.[\p{L}\p{N}][\p{L}\p{N}\p{M}_+-]*$/u.test(link.pathText) + return ( + (!requireSeparator || /[\\/]/.test(link.pathText)) && + (hasRootedPrefix || + hasLineSuffix || + (requireSeparator ? hasPathExtension : hasAlphabeticExtension)) && + routeNativeChatHref(link.displayText).kind === 'file' + ) +} + +const SAFE_LEADING_BOUNDARY_PATTERN = /[\s([{'",;=]/ +const SAFE_TRAILING_BOUNDARY_PATTERN = /[\s)\]}>'",;.:。!?,、;:]/ +const SENTENCE_PATH_PUNCTUATION_PATTERN = + /\.[\p{L}\p{N}][\p{L}\p{N}\p{M}_+-]*([!?—。!?,、;:])/gu +const QUOTED_TEXT_PATTERN = /"([^"\r\n]+)"|'([^"'\r\n]+)'/gu +const MAX_DASHED_PROSE_WORD_LENGTH = 32 + +function hasBoundedProseAfterDash(value: string, startIndex: number): boolean { + const endIndex = Math.min(value.length, startIndex + MAX_DASHED_PROSE_WORD_LENGTH) + for (let index = startIndex; index < endIndex; index += 1) { + const char = value[index] + if (!char || SAFE_TRAILING_BOUNDARY_PATTERN.test(char)) { + return true + } + if (char === '/' || char === '\\') { + return false + } + } + return endIndex === value.length +} + +function isSafeTrailingBoundary(value: string, endIndex: number): boolean { + const boundary = value[endIndex] + if (boundary === undefined || SAFE_TRAILING_BOUNDARY_PATTERN.test(boundary)) { + return true + } + if (boundary === '!' || boundary === '?') { + const next = value[endIndex + 1] + return next === undefined || SAFE_TRAILING_BOUNDARY_PATTERN.test(next) + } + if (boundary === '—') { + return hasBoundedProseAfterDash(value, endIndex + 1) + } + return false +} + +function hasPartialPathBoundary(value: string, link: ParsedTerminalFileLink): boolean { + const before = value[link.startIndex - 1] + return ( + (before !== undefined && !SAFE_LEADING_BOUNDARY_PATTERN.test(before)) || + !isSafeTrailingBoundary(value, link.endIndex) + ) +} + +function createFileLinkNode(value: string, child: MarkdownNode): MarkdownNode { + return { + type: 'link', + url: createNativeChatFileHref(value), + children: [child] + } +} + +// Why: the terminal extractor spans "src/a.ts and src/b.ts" as one spaced path. +// An unrooted span holding a bare word or several linkable tokens is prose +// joining paths, so link the tokens on their own; a spaced folder name keeps +// every token path-shaped and stays one link. +function splitProseJoinedLinks(link: ParsedTerminalFileLink): ParsedTerminalFileLink[] { + if (ROOTED_PATH_PREFIX_PATTERN.test(link.pathText)) { + return [link] + } + const tokens = Array.from(link.displayText.matchAll(/\S+/g)) + const tokenLinks: ParsedTerminalFileLink[] = [] + for (const match of tokens) { + const token = match[0] + const exactLink = extractTerminalFileLinks(token).find( + (candidate) => candidate.startIndex === 0 && candidate.endIndex === token.length + ) + if (exactLink && isLinkifiableFile(exactLink, true)) { + const startIndex = link.startIndex + (match.index ?? 0) + tokenLinks.push({ ...exactLink, startIndex, endIndex: startIndex + token.length }) + } + } + const hasBareWord = tokens.some((match) => !/[\\/.]/.test(match[0])) + return hasBareWord || tokenLinks.length > 1 ? tokenLinks : [link] +} + +function splitTextSegment(value: string): MarkdownNode[] { + const links = extractTerminalFileLinks(value) + .filter((link) => !hasPartialPathBoundary(value, link)) + .filter((link) => isLinkifiableFile(link, true)) + .flatMap(splitProseJoinedLinks) + if (links.length === 0) { + return [{ type: 'text', value }] + } + + const children: MarkdownNode[] = [] + let cursor = 0 + for (const link of links) { + if (link.startIndex < cursor) { + continue + } + if (link.startIndex > cursor) { + children.push({ type: 'text', value: value.slice(cursor, link.startIndex) }) + } + children.push(createFileLinkNode(link.displayText, { type: 'text', value: link.displayText })) + cursor = link.endIndex + } + if (cursor < value.length) { + children.push({ type: 'text', value: value.slice(cursor) }) + } + return children +} + +function splitUnquotedText(value: string): MarkdownNode[] { + const children: MarkdownNode[] = [] + let cursor = 0 + for (const match of value.matchAll(SENTENCE_PATH_PUNCTUATION_PATTERN)) { + const punctuationIndex = (match.index ?? 0) + match[0].length - 1 + if (!isSafeTrailingBoundary(value, punctuationIndex)) { + continue + } + children.push(...splitTextSegment(value.slice(cursor, punctuationIndex))) + children.push({ type: 'text', value: value[punctuationIndex] }) + cursor = punctuationIndex + 1 + } + if (cursor === 0) { + return splitTextSegment(value) + } + children.push(...splitTextSegment(value.slice(cursor))) + return children +} + +function exactFileLink(value: string, allowSpacedRelative: boolean): ParsedTerminalFileLink | null { + const exactLink = extractTerminalFileLinks(value).find( + (link) => link.startIndex === 0 && link.endIndex === value.length + ) + if (exactLink && isLinkifiableFile(exactLink, false)) { + return exactLink + } + if (!allowSpacedRelative || !/\s/.test(value)) { + return null + } + const parsed = parseFileLinkLocation(value) + if (!parsed) { + return null + } + const hasPathShape = + ROOTED_PATH_PREFIX_PATTERN.test(parsed.pathText) || + /[\\/]/.test(parsed.pathText) || + /\.[\p{L}][\p{L}\p{N}\p{M}_+-]*$/u.test(parsed.pathText) + if (!hasPathShape) { + return null + } + const explicitLink = { + ...parsed, + startIndex: 0, + endIndex: value.length, + displayText: value + } + return isLinkifiableFile(explicitLink, false) ? explicitLink : null +} + +function splitTextNode(value: string): MarkdownNode[] { + const children: MarkdownNode[] = [] + let cursor = 0 + for (const match of value.matchAll(QUOTED_TEXT_PATTERN)) { + const content = match[1] ?? match[2] + if (!content || !exactFileLink(content, true)) { + continue + } + const matchIndex = match.index ?? 0 + const quote = match[0][0] + children.push(...splitUnquotedText(value.slice(cursor, matchIndex))) + children.push({ type: 'text', value: quote }) + children.push(createFileLinkNode(content, { type: 'text', value: content })) + children.push({ type: 'text', value: quote }) + cursor = matchIndex + match[0].length + } + if (cursor === 0) { + return splitUnquotedText(value) + } + children.push(...splitUnquotedText(value.slice(cursor))) + return children +} + +function inlineCodeFileLink(node: MarkdownNode): MarkdownNode | null { + const value = node.value?.trim() + if (!value) { + return null + } + return exactFileLink(value, true) ? createFileLinkNode(value, node) : null +} + +function transformFileLinks(node: MarkdownNode): void { + if (node.type === 'link') { + if (node.url && routeNativeChatHref(node.url).kind === 'file') { + node.url = createNativeChatFileHref(node.url) + } + return + } + if (!node.children || node.type === 'image') { + return + } + + const children: MarkdownNode[] = [] + for (const child of node.children) { + if (child.type === 'text' && child.value !== undefined) { + children.push(...splitTextNode(child.value)) + continue + } + if (child.type === 'inlineCode') { + children.push(inlineCodeFileLink(child) ?? child) + continue + } + transformFileLinks(child) + children.push(child) + } + node.children = children +} + +export function remarkNativeChatFileLinks(): (tree: MarkdownNode) => void { + return (tree) => transformFileLinks(tree) +} diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx index 259037afea9..da2a3f02133 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx @@ -76,7 +76,6 @@ export function TerminalPaneNativeChatPortal({ agent={structuredChatAgent} isVisible={isRendererVisible} target={structuredChatTarget} - allowFileUriLinks contextMenuActions={contextMenuActions} orchestrationDispatchStatus={chatPaneDispatchStatus} /> diff --git a/src/renderer/src/lib/terminal-links.test.ts b/src/renderer/src/lib/terminal-links.test.ts index 66ff8d870a8..d16e3d3b0cc 100644 --- a/src/renderer/src/lib/terminal-links.test.ts +++ b/src/renderer/src/lib/terminal-links.test.ts @@ -115,6 +115,15 @@ describe('terminal path helpers', () => { }) describe('extractTerminalFileLinks local path tokens', () => { + it('keeps Unicode path segments in the detected range', () => { + expect(extractTerminalFileLinks('/tmp/报告.html').map((link) => link.displayText)).toEqual([ + '/tmp/报告.html' + ]) + expect( + extractTerminalFileLinks('docs/café/report.pdf').map((link) => link.displayText) + ).toEqual(['docs/café/report.pdf']) + }) + it('detects tilde-prefixed POSIX paths', () => { const links = extractTerminalFileLinks('~/Documents/Path/file_name') expect(links).toHaveLength(1) @@ -215,6 +224,15 @@ describe('terminal path helpers', () => { expect(links).toHaveLength(20_000) expect(links[0].pathText).toBe('/tmp/Foo Bar/file') }, 5_000) + + it('keeps extension-heavy assistant prose on a bounded scan path', () => { + const filenames = Array.from({ length: 20_000 }, (_, index) => `report-${index}.txt`) + const links = extractTerminalFileLinks(`/tmp/root/${filenames.join(' ')}`) + + expect(links).toHaveLength(2) + expect(links[0].pathText).toBe(`/tmp/root/${filenames.slice(0, -1).join(' ')}`) + expect(links.at(-1)?.pathText).toBe('report-19999.txt') + }) }) it('supports Windows cwd resolution for terminal file links', () => { diff --git a/src/renderer/src/lib/terminal-links.ts b/src/renderer/src/lib/terminal-links.ts index 729f2d839e7..970c6d9208c 100644 --- a/src/renderer/src/lib/terminal-links.ts +++ b/src/renderer/src/lib/terminal-links.ts @@ -34,7 +34,7 @@ export type ResolvedTerminalFileLink = Pick 1) { + while (nextPathStart && nextPathStart.index + nextPathStart[0].length <= end) { + pathStartCount += 1 + nextPathStart = pathStartPattern.exec(range.text) + } + if (pathStartCount > 1) { continue } if ( @@ -150,15 +157,6 @@ function trimSpacedPathTrailingProse( } } -function countPathStarts(text: string): number { - let count = 0 - for (const match of text.matchAll(/(?:^|\s)(?:~[\\/]|[\\/]|\.{1,2}[\\/]|[A-Za-z]:[\\/])/g)) { - void match - count += 1 - } - return count -} - function trimTrailingWhitespace( range: DetectedTerminalFileLinkRange ): DetectedTerminalFileLinkRange { diff --git a/src/shared/native-chat-href-routing.test.ts b/src/shared/native-chat-href-routing.test.ts index 1b7c276ffd3..5f6e7ad4595 100644 --- a/src/shared/native-chat-href-routing.test.ts +++ b/src/shared/native-chat-href-routing.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { routeNativeChatHref } from './native-chat-href-routing' +import { createNativeChatFileHref, routeNativeChatHref } from './native-chat-href-routing' describe('routeNativeChatHref', () => { it('classifies web and mail links', () => { @@ -47,6 +47,35 @@ describe('routeNativeChatHref', () => { }) }) + it('routes encoded renderer file targets without treating Windows drives as schemes', () => { + expect( + routeNativeChatHref(createNativeChatFileHref(String.raw`C:\repo\report.docx:12`)) + ).toEqual({ + kind: 'file', + pathText: String.raw`C:\repo\report.docx:12`, + line: null + }) + expect(routeNativeChatHref(createNativeChatFileHref('/tmp/report.html'))).toEqual({ + kind: 'file', + pathText: '/tmp/report.html', + line: null + }) + }) + + it('bounds nested renderer file target decoding', () => { + let href = '/tmp/report.html' + for (let depth = 0; depth < 4; depth += 1) { + href = createNativeChatFileHref(` ${href}`) + } + expect(routeNativeChatHref(href)).toEqual({ + kind: 'file', + pathText: '/tmp/report.html', + line: null + }) + + expect(routeNativeChatHref(createNativeChatFileHref(` ${href}`))).toEqual({ kind: 'none' }) + }) + it('drops anchors, unknown schemes, malformed file URIs, and empty hrefs', () => { expect(routeNativeChatHref('#section')).toEqual({ kind: 'none' }) expect(routeNativeChatHref(undefined)).toEqual({ kind: 'none' }) diff --git a/src/shared/native-chat-href-routing.ts b/src/shared/native-chat-href-routing.ts index 005ab957f65..edf36565dab 100644 --- a/src/shared/native-chat-href-routing.ts +++ b/src/shared/native-chat-href-routing.ts @@ -8,6 +8,24 @@ export type NativeChatHrefRoute = const WEB_SCHEME_PATTERN = /^(?:https?|mailto):/i const SCHEME_PATTERN = /^[A-Za-z][A-Za-z0-9+.-]*:/ +export const NATIVE_CHAT_FILE_HREF_PREFIX = '#orca-native-chat-file=' +const MAX_NATIVE_CHAT_FILE_HREF_DECODES = 4 + +export function createNativeChatFileHref(pathText: string): string { + return `${NATIVE_CHAT_FILE_HREF_PREFIX}${encodeURIComponent(pathText)}` +} + +function decodeNativeChatFileHref(href: string): string | null { + if (!href.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX)) { + return null + } + try { + const decoded = decodeURIComponent(href.slice(NATIVE_CHAT_FILE_HREF_PREFIX.length)) + return decoded && !decoded.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX) ? decoded : null + } catch { + return null + } +} function parseLineFragment(hash: string): number | null { if (!hash) { @@ -45,8 +63,21 @@ function maybeDecodeHrefPath(value: string): string { } export function routeNativeChatHref(href: string | null | undefined): NativeChatHrefRoute { - const trimmed = href?.trim() - if (!trimmed || trimmed.startsWith('#')) { + let trimmed = href?.trim() + if (!trimmed) { + return { kind: 'none' } + } + for (let depth = 0; depth < MAX_NATIVE_CHAT_FILE_HREF_DECODES; depth += 1) { + const encodedFileHref = decodeNativeChatFileHref(trimmed) + if (!encodedFileHref) { + break + } + trimmed = encodedFileHref.trim() + } + if (!trimmed || trimmed.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX)) { + return { kind: 'none' } + } + if (trimmed.startsWith('#')) { return { kind: 'none' } } if (WEB_SCHEME_PATTERN.test(trimmed)) { From cb7f7dd11a3948e9022bb64170335c7e313263aa Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Fri, 4 Sep 2026 20:08:39 -0700 Subject: [PATCH 02/26] fix(native-chat): tell old mobile builds why a structured chat is missing (#18756) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(native-chat): tell old mobile builds why a structured chat is missing A structured native chat started on desktop was simply absent on a paired phone running any shipped App Store build. The host strips every `agent-session` tab from a client that does not advertise `agent-session.structured.v1`, and no released mobile build advertises it — so the chat had no representation at all and no way to explain itself. Keep the row and retitle it instead of deleting it. The shipped client does not filter unknown tab types and renders whatever title the host sends, so an old build now shows the chat's slot with a title naming the fix. Nothing is removed, so the tab order, groups and layout it belonged to are left intact. The prompt is keyed on the capability for that specific agent, not on the combined policy boolean: a capable phone whose desktop simply has the experiment off would otherwise be told to take an update that cannot help it. Claude rows are prompted too — mobile cannot render them yet and a later build can, so the message is true for that client as well. Restore is no longer gated on the caller's capability. It stayed gated on the host setting, which is what decides whether there is anything to reach at all, but gating on capability left an old client with nothing to project after a desktop restart: neither the chat nor the prompt. Tab titles are capped at 128px on one line in every shipped build, so the string is sized for ~15 characters rather than a sentence. Prompted rows are visible rows, so the host now permits all five session-tab mutations on them, close included. That is intended: a mobile close runs the same teardown as the desktop's own Close button. * fix(native-chat): keep fallback tabs safe and truthful --------- Co-authored-by: Merge Sim --- ...ion-tab-agent-capability-mutations.test.ts | 44 ++++++- ...ession-tab-agent-status-projection.test.ts | 109 ++++++++++++++--- .../session-tab-agent-status-projection.ts | 110 ++++++++++++++++-- .../rpc/methods/session-tab-close-methods.ts | 25 +++- .../session-tabs-structured-restore.test.ts | 98 ++++++++++++++++ .../runtime/rpc/methods/session-tabs.test.ts | 47 +------- .../methods/structured-agent-session-gate.ts | 7 +- .../rpc/methods/structured-agent-session.ts | 5 +- .../methods/structured-session-tab-restore.ts | 21 +++- src/shared/protocol-version.ts | 11 +- ...ss-version-agent-session-wire.unit.test.ts | 11 +- 11 files changed, 393 insertions(+), 95 deletions(-) create mode 100644 src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts diff --git a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts index a2a93533b6f..183f981ccee 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts @@ -50,6 +50,8 @@ const METHODS = [ } ] as const +const DESTRUCTIVE_METHOD_NAMES = new Set(['session.tabs.close', 'session.tabs.closeLifecycle']) + describe('session tab structured capability mutations', () => { for (const method of METHODS) { it(`rejects ${method.name} when the structured row is hidden`, async () => { @@ -88,6 +90,38 @@ describe('session tab structured capability mutations', () => { }) } + for (const method of METHODS) { + const expectedToAllowPromptedRow = !DESTRUCTIVE_METHOD_NAMES.has(method.name) + + it(`${expectedToAllowPromptedRow ? 'allows' : 'rejects'} ${method.name} on a row an old mobile client was prompted to update`, async () => { + const { calls, dispatch } = createFixture([], { + clientKind: 'mobile', + structuredNativeChatEnabled: true + }) + + const response = await dispatch(method.name, method.params('codex-session')) + + expect(response.ok).toBe(expectedToAllowPromptedRow) + expect(calls[method.runtimeMethod as keyof typeof calls]).toHaveBeenCalledTimes( + expectedToAllowPromptedRow ? 1 : 0 + ) + }) + + it(`${expectedToAllowPromptedRow ? 'allows' : 'rejects'} ${method.name} on a prompted Claude row for a mobile client without the Claude capability`, async () => { + const { calls, dispatch } = createFixture([STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], { + clientKind: 'mobile', + structuredNativeChatEnabled: true + }) + + const response = await dispatch(method.name, method.params('claude-session')) + + expect(response.ok).toBe(expectedToAllowPromptedRow) + expect(calls[method.runtimeMethod as keyof typeof calls]).toHaveBeenCalledTimes( + expectedToAllowPromptedRow ? 1 : 0 + ) + }) + } + it.each(['session.tabs.close', 'session.tabs.closeLifecycle'] as const)( 'allows capable mobile clients to close structured tabs when the experiment is enabled (%s)', async (method) => { @@ -131,7 +165,10 @@ describe('session tab structured capability mutations', () => { ) }) -function createFixture(capabilities: RuntimeCapability[]) { +function createFixture( + capabilities: RuntimeCapability[], + options: { clientKind?: 'mobile' | 'runtime'; structuredNativeChatEnabled?: boolean } = {} +) { const snapshot = agentSnapshot() const calls = { closeMobileSessionTab: vi.fn().mockResolvedValue({ closed: true }), @@ -142,11 +179,14 @@ function createFixture(capabilities: RuntimeCapability[]) { const runtime = { getRuntimeId: () => 'test-runtime', listMobileSessionTabs: vi.fn().mockResolvedValue(snapshot), + getClientSettings: () => ({ + experimentalStructuredNativeChat: options.structuredNativeChatEnabled === true + }), ...calls } as unknown as OrcaRuntimeService const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) const context: RpcDispatchStreamingOptions = { - clientKind: 'runtime', + clientKind: options.clientKind ?? 'runtime', pairedDeviceId: 'paired-client', clientCapabilities: capabilities } diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts index 7128f756d6e..cf68f7739c0 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts @@ -5,7 +5,11 @@ import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { RuntimeMobileSessionTabsSnapshot } from '../../../../shared/runtime-types' -import { projectSessionTabAgentStatus } from './session-tab-agent-status-projection' +import { + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE, + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + projectSessionTabAgentStatus +} from './session-tab-agent-status-projection' function makeSnapshot(sessionBoundary: boolean): RuntimeMobileSessionTabsSnapshot { return { @@ -160,23 +164,96 @@ describe('projectSessionTabAgentStatus', () => { CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ] - it.each([ - ['mobile', 'mobile' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]], - ['runtime', 'runtime' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]] - ])( - 'withholds Claude rows from a paired %s client that never negotiated them', - (_name, clientKind, capabilities) => { - const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + it('withholds Claude rows from a paired runtime client that never negotiated them', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'runtime', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) - expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) - // A row pruned from `tabs` but left in the layout is its own dead tab. - expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) - expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) - expect(projected.activeGroupId).toBe('group-a') - expect(projected.activeTabId).toBe('agent-session:codex') - expect(projected.activeTabType).toBe('agent-session') + expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) + // A row pruned from `tabs` but left in the layout is its own dead tab. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) + expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) + expect(projected.activeGroupId).toBe('group-a') + expect(projected.activeTabId).toBe('agent-session:codex') + expect(projected.activeTabType).toBe('agent-session') + }) + + it('uses a desktop fallback for an unsupported Claude row instead of withholding it', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'mobile', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + // The row survives so the chat the desktop shows is not simply absent on the phone. + expect(projected.tabs.map((tab) => tab.id)).toEqual([ + 'agent-session:codex', + 'agent-session:claude' + ]) + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + 'Codex Chat', + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + // Nothing is removed, so the layout it belonged to is untouched. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a', 'group-b']) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + expect(projected.activeTabId).toBe('agent-session:codex') + }) + + it('projects agent-specific fallback titles for a mobile client with no capabilities', () => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', [], true) + + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + }) + + it('does not treat the Claude capability as a substitute for the base structured capability', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'mobile', + [CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + }) + + it('shows both real titles once mobile negotiates Claude', () => { + expect(projectSessionTabAgentStatus(claudeSnapshot, 'mobile', structuredMobile, true)).toBe( + claudeSnapshot + ) + }) + + // Why: updating cannot reveal a chat the desktop is not serving, so the prompt would lie. + it('withholds rather than prompts when the desktop experiment is off', () => { + for (const capabilities of [ + [], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + structuredMobile + ]) { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', capabilities, false) + expect(projected.tabs).toEqual([]) } - ) + }) + + it('never emits an empty structured tab title', () => { + for (const capabilities of [[], [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]]) { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', capabilities, true) + for (const tab of projected.tabs) { + expect(tab.title.length).toBeGreaterThan(0) + } + } + }) it.each([ ['mobile', 'mobile' as const, structuredMobile], diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts index 0e0d9c716a5..4496fdc5435 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts @@ -1,6 +1,7 @@ import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' import type { @@ -13,6 +14,43 @@ import { structuredNativeChatProjectionEnabled } from './structured-agent-sessio type SessionTabsPayload = RuntimeMobileSessionTabsResult | RuntimeMobileSessionTabsSnapshot +/** Capped at 128px / one line in every shipped mobile build, so ~15-18 characters render. */ +export const STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE = 'Update to view' +export const CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE = 'Open on desktop' + +function clientCanRenderStructuredAgentSessionTab( + tab: RuntimeMobileSessionAgentTab, + clientCapabilities: readonly RuntimeCapability[] | undefined +): boolean { + if (!clientCapabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY)) { + return false + } + return ( + tab.agent === 'codex' || + clientCapabilities.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) +} + +function resolveMobileStructuredChatFallbackTitle( + tab: RuntimeMobileSessionAgentTab, + args: { + clientKind: 'mobile' | 'runtime' | undefined + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled?: boolean + } +): string | null { + if ( + args.clientKind !== 'mobile' || + args.structuredNativeChatEnabled !== true || + clientCanRenderStructuredAgentSessionTab(tab, args.clientCapabilities) + ) { + return null + } + return tab.agent === 'claude' + ? CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + : STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE +} + export function projectSessionTabAgentStatus( payload: TPayload, clientKind: 'mobile' | 'runtime' | undefined, @@ -24,16 +62,27 @@ export function projectSessionTabAgentStatus true) - // Why: a paired client renders only codex structured tabs unless it says otherwise - // (mobile's resolveMobileNativeChat returns null for every other agent), so an - // ungated row would list and select into a pane that shows neither chat nor terminal. - if ( - structuredVisible && - clientKind !== undefined && - !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) - ) { - projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') + let projected: TPayload + if (clientKind === 'mobile' && structuredNativeChatEnabled === true) { + // Why: deleting the row left the user hunting for a chat the desktop says exists; the row + // survives with a title naming the fix. Nothing is removed, so no group/layout repair applies. + projected = projectUnsupportedAgentSessionTabTitles(payload, { + clientKind, + clientCapabilities, + structuredNativeChatEnabled + }) + } else { + projected = structuredVisible ? payload : projectAgentSessionTabsOut(payload, () => true) + // Why: a paired client renders only codex structured tabs unless it says otherwise + // (mobile's resolveMobileNativeChat returns null for every other agent), so an + // ungated row would list and select into a pane that shows neither chat nor terminal. + if ( + structuredVisible && + clientKind !== undefined && + !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) { + projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') + } } // Why: only paired runtimes have legacy `done` completion side effects; mobile must keep its row without changing the exact v2 auth shape. if ( @@ -55,6 +104,47 @@ export function projectSessionTabAgentStatus( + payload: TPayload, + args: { + clientKind: 'mobile' + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled: true + } +): TPayload { + let changed = false + const tabs = payload.tabs.map((tab) => { + if (tab.type !== 'agent-session') { + return tab + } + const title = resolveMobileStructuredChatFallbackTitle(tab, args) + if (title === null) { + return tab + } + changed = true + return { ...tab, title } + }) + return changed ? ({ ...payload, tabs } as TPayload) : payload +} + +export function assertAgentSessionTabDestructiveMutationSupported( + payload: SessionTabsPayload, + tabId: string, + clientKind: 'mobile' | 'runtime' | undefined, + clientCapabilities: readonly RuntimeCapability[] | undefined +): void { + if (clientKind === undefined) { + return + } + const tab = payload.tabs.find((candidate) => candidate.id === tabId) + if ( + tab?.type === 'agent-session' && + !clientCanRenderStructuredAgentSessionTab(tab, clientCapabilities) + ) { + throw new Error('structured_agent_session_unsupported') + } +} + function projectAgentSessionTabsOut( payload: TPayload, shouldHide: (tab: RuntimeMobileSessionAgentTab) => boolean diff --git a/src/main/runtime/rpc/methods/session-tab-close-methods.ts b/src/main/runtime/rpc/methods/session-tab-close-methods.ts index bd60ecd6ddf..361ba8e4c51 100644 --- a/src/main/runtime/rpc/methods/session-tab-close-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-close-methods.ts @@ -3,6 +3,7 @@ import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/ import { defineMethod, type RpcAnyMethod } from '../core' import { CloseLifecycleTab, CloseTab } from './session-tabs-schemas' import { assertProjectedSessionTabVisible } from './session-tab-browser-placement-projection' +import { assertAgentSessionTabDestructiveMutationSupported } from './session-tab-agent-status-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' @@ -12,8 +13,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ params: CloseTab, handler: async (params, context) => { if (context.clientKind) { + const raw = await context.runtime.listMobileSessionTabs( + params.worktree, + context.pairedDeviceId + ) const visible = projectSessionTabsForClient( - await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), + raw, context.clientKind, context.clientCapabilities, context.clientKind === 'mobile' @@ -21,6 +26,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) + assertAgentSessionTabDestructiveMutationSupported( + raw, + params.tabId, + context.clientKind, + context.clientCapabilities + ) } const requiresIntent = context.clientKind === undefined || @@ -81,8 +92,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ params: CloseLifecycleTab, handler: async (params, context) => { if (context.clientKind) { + const raw = await context.runtime.listMobileSessionTabs( + params.worktree, + context.pairedDeviceId + ) const visible = projectSessionTabsForClient( - await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), + raw, context.clientKind, context.clientCapabilities, context.clientKind === 'mobile' @@ -90,6 +105,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) + assertAgentSessionTabDestructiveMutationSupported( + raw, + params.tabId, + context.clientKind, + context.clientCapabilities + ) } return withSpan( 'runtime.session-tabs.close-lifecycle', diff --git a/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts b/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts new file mode 100644 index 00000000000..083f334285e --- /dev/null +++ b/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import { RpcDispatcher } from '../dispatcher' +import type { RpcRequest } from '../core' +import type { OrcaRuntimeService } from '../../orca-runtime' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import { SESSION_TAB_METHODS } from './session-tabs' + +function makeRequest(method: string, params?: unknown): RpcRequest { + return { id: 'req-1', authToken: 'tok', method, params } +} + +describe('session tab structured restore gating', () => { + it('does not restore structured tabs for mobile while the host setting is off', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() + }) + + // Why: an old build has no capability to advertise, and skipping the restore left it with + // nothing to project after a desktop restart — neither the chat nor its fallback row. + it('restores structured tabs for a mobile client that advertises no capability', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { clientKind: 'mobile', clientCapabilities: [] } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) + + it('restores structured tabs for mobile once the setting is present', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) +}) + +function visibleSnapshot() { + return { + worktree: 'wt-1', + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: 'group-1', + activeTabId: 'tab-1::leaf-1', + activeTabType: 'terminal' as const, + tabGroups: [{ id: 'group-1', activeTabId: 'tab-1', tabOrder: ['tab-1'] }], + tabs: [ + { + type: 'terminal' as const, + id: 'tab-1::leaf-1', + parentTabId: 'tab-1', + leafId: 'leaf-1', + title: 'Terminal', + status: 'ready' as const, + terminal: 'pty-1', + isActive: true + } + ] + } +} diff --git a/src/main/runtime/rpc/methods/session-tabs.test.ts b/src/main/runtime/rpc/methods/session-tabs.test.ts index 29131869fa0..be61fc55edf 100644 --- a/src/main/runtime/rpc/methods/session-tabs.test.ts +++ b/src/main/runtime/rpc/methods/session-tabs.test.ts @@ -2,10 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' import type { RpcRequest } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' -import { - SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY, - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY -} from '../../../../shared/protocol-version' +import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { SESSION_TAB_METHODS } from './session-tabs' function makeRequest(method: string, params?: unknown): RpcRequest { @@ -13,48 +10,6 @@ function makeRequest(method: string, params?: unknown): RpcRequest { } describe('session tab RPC methods', () => { - it('does not restore structured tabs for mobile while the host setting is off', async () => { - const runtime = { - getRuntimeId: () => 'test-runtime', - getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), - restoreStructuredAgentSessionTabs: vi.fn(), - listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) - } as unknown as OrcaRuntimeService - const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) - - const response = await dispatcher.dispatch( - makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), - { - clientKind: 'mobile', - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] - } - ) - - expect(response.ok).toBe(true) - expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() - }) - - it('restores structured tabs for mobile only after capability and setting are present', async () => { - const runtime = { - getRuntimeId: () => 'test-runtime', - getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), - restoreStructuredAgentSessionTabs: vi.fn(), - listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) - } as unknown as OrcaRuntimeService - const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) - - const response = await dispatcher.dispatch( - makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), - { - clientKind: 'mobile', - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] - } - ) - - expect(response.ok).toBe(true) - expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) - }) - it('routes mobile-only activation without notifying desktop clients', async () => { const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts index 33018c21f4d..e15544b2d24 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts @@ -2,8 +2,11 @@ // // Shared by every structured method file so one gate governs the whole surface: a client that does // not advertise `agent-session.structured.v1` is told the surface does not exist rather than being -// handed a session it cannot render or drive — and, just as importantly, cannot make the host EXIST -// by calling into it, which is an observable side effect. +// handed the session journal or mutation surface. +// +// This gate no longer implies such a client cannot make the host exist: session-tab restore runs +// for old mobile clients while structured chat is enabled so they receive a fallback row, and that +// path constructs the host. `agentSession.*` stays refused either way, which is what this gate is for. import { getStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index 3b18f6b0ef1..abe196bd636 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -2,9 +2,8 @@ // // Every method here is gated on the client advertising // `agent-session.structured.v1`. A client that does not is told the surface does -// not exist rather than being handed a session it cannot render or drive; that -// is the whole visibility rule, because nothing else on the runtime publishes a -// structured session. +// not exist rather than receiving the journal or mutation surface. Session-tab +// inventory may expose only a metadata placeholder for an incapable mobile client. import { agentSessionFingerprintConflict, diff --git a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts index 4333713a445..f9efa46d16d 100644 --- a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts +++ b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts @@ -1,13 +1,24 @@ import type { RpcContext } from '../core' -import { supportsStructuredAgentSessions } from './structured-agent-session-policy' +import { + isStructuredNativeChatEnabled, + supportsStructuredAgentSessions +} from './structured-agent-session-policy' +/** Republishes structured tabs into the host's own snapshot map. + * + * Mobile is gated on the host setting alone, NOT on the client's capability: an old build is + * shown a fallback prompt in place of each chat, and gating on capability left it with nothing to + * project after a desktop restart — no chat and no prompt. The setting still gates it, because + * with structured chat off there is nothing for any mobile client to reach. Restoring spawns no + * provider child for a cleanly closed session. */ export async function restoreStructuredTabsIfSupported( context: Pick ): Promise { - if ( - supportsStructuredAgentSessions(context) && - typeof context.runtime.restoreStructuredAgentSessionTabs === 'function' - ) { + const shouldRestore = + context.clientKind === 'mobile' + ? isStructuredNativeChatEnabled(context.runtime) + : supportsStructuredAgentSessions(context) + if (shouldRestore && typeof context.runtime.restoreStructuredAgentSessionTabs === 'function') { await context.runtime.restoreStructuredAgentSessionTabs() } } diff --git a/src/shared/protocol-version.ts b/src/shared/protocol-version.ts index d78c6223c99..76e1252640a 100644 --- a/src/shared/protocol-version.ts +++ b/src/shared/protocol-version.ts @@ -121,13 +121,12 @@ export const AGENT_SESSION_HOST_AUTHORITY_RUNTIME_CAPABILITY = 'agent-session.host-authority.v1' as const export const AGENT_SESSION_OMP_RESUME_PATH_RUNTIME_CAPABILITY = 'agent-session.omp-resume-path.v1' as const -// Why: structured sessions are journal-backed, not PTY-backed, so a client that -// cannot read them must not see them at all — it would render an agent tab it -// can neither display nor drive. The host also refuses every agentSession.* -// method from a connection that does not advertise this. +// Why: structured sessions are journal-backed, not PTY-backed, so an incapable client must not +// receive their journal or drive their lifecycle. Mobile may receive a metadata-only placeholder; +// the host still refuses agentSession.* methods and destructive tab mutations without capability. export const STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = 'agent-session.structured.v1' as const -// Why: mobile clients advertise Claude-structured session support during E2EE -// pairing so the desktop can keep structured-specific affordances enabled. +// Why: paired clients advertise Claude-structured support so the host can gate its agent-specific +// journal and lifecycle surfaces independently from Codex support. export const CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = 'agent-session.structured.claude.v1' as const // Why: paired structured clients explicitly hold every visible session surface, allowing the host diff --git a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts index 2ad8bd0647f..a2a64897fcb 100644 --- a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts +++ b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts @@ -2,9 +2,14 @@ // way the terminal wire harness is: current code against a real published release. // // Three skews matter here, and none can be checked from one build alone — an old -// client must not be shown a session it cannot render, a new client must find an -// old host's missing surface cleanly, and a client's cursor must survive the host -// process that minted it. +// client must not receive a journal-backed RPC surface it cannot read, a new client +// must find an old host's missing surface cleanly, and a client's cursor must survive +// the host process that minted it. +// +// The session-tabs projection may keep a metadata-only row for an incapable mobile client so the +// chat is not simply absent on the phone. Every `agentSession.*` method and destructive close stays +// refused, which is what the tests below pin; the row-level behaviour is pinned in +// src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts. import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' From 974acc901c6921e02e34053449c58859cb86b24d Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:13:55 -0400 Subject: [PATCH 03/26] fix(relay-ops): retry freshness-only preflight failures on the first same-cap wave too (#18778) --- ...d-deploy-relay-production-same-cap-job.yml | 7 ++-- .../src/incident-live-preflight-cli.test.ts | 33 +++++++++++++++++++ .../src/incident-live-preflight-cli.ts | 6 +++- .../relay-regional-rehome-workflow.test.mjs | 6 +++- 4 files changed, 47 insertions(+), 5 deletions(-) diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index c4eea80933e..d5c8134934b 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -181,11 +181,12 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }} run: | - RETRY_ARGS=() - if test "${WAVE_INDEX}" != 0; then RETRY_ARGS=(--retry-freshness); fi + # Freshness-only failures are publish lag, not health, on every wave + # including the first; the CLI still caps the retry at the wave's + # evidence-age budget, so this cannot mutate on aged evidence. pnpm incident:relay-preflight -- \ --state-file "${OUTPUT_DIRECTORY}/relay-${MONITOR_RUN_ID}-dry-run.state.json" \ - --wave-index "${WAVE_INDEX}" "${RETRY_ARGS[@]}" + --wave-index "${WAVE_INDEX}" --retry-freshness - name: Require durable rehome disabled and exact selector env: diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts index 18ee4495078..3252ec7645e 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts @@ -331,6 +331,39 @@ describe('relay incident live preflight', () => { expect(wait).toHaveBeenNthCalledWith(2, 15_000) }) + it('retries a first-wave stale sample and passes on the fresh one', async () => { + const stale = sample() + stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = + new Date(now - 180_001).toISOString() + const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample()) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', '0', '--retry-freshness'], + { now: () => now, collect, wait } + )).resolves.toBeUndefined() + expect(collect).toHaveBeenCalledTimes(2) + expect(wait).toHaveBeenCalledOnce() + }) + + it('stops retrying when the next wait would exceed the evidence-age bound', async () => { + const completedAt = now - 290_000 + const stale = sample() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + const collect = vi.fn(async () => stale) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('strict', { + startedAt: new Date(completedAt - 17 * 60_000).toISOString(), + windowStartedAt: new Date(completedAt - 16 * 60_000).toISOString(), + lastSampleAt: new Date(completedAt - 30_000).toISOString(), + completedAt: new Date(completedAt).toISOString() + }), '--retry-freshness'], + { now: () => now, collect, wait } + )).rejects.toThrow('cloud-monitoring/source_stale') + expect(collect).toHaveBeenCalledOnce() + expect(wait).not.toHaveBeenCalled() + }) + it('does not retry a threshold failure', async () => { const unhealthy = sample() unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts index e82627a3e80..fc2c3a99751 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts @@ -173,7 +173,11 @@ export async function runIncidentLivePreflight( const freshnessOnly = evaluation.failures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code) ) - if (!freshnessOnly || attempt === attempts) { + // Waiting must never carry the mutation past the same evidence-age bound + // the entry check enforces, so the wave budget also caps the retry window. + const budgetExhausted = + now() + FRESHNESS_RETRY_INTERVAL_MS - completedAt > maxEvidenceAgeMs + if (!freshnessOnly || attempt === attempts || budgetExhausted) { throw new Error( `relay live preflight failed: ${evaluation.failures .map((failure) => `${failure.source}/${failure.code}`) diff --git a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs index e31403f6dd6..a9e92d6cd57 100644 --- a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs +++ b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs @@ -92,7 +92,11 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => { // age checks must scale by wave or cell_2+ can never pass; the bound's // per-wave step is the cell job timeout, so the two must move together. assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/) - assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/) + // Wave 0 must retry freshness-only failures too: one Cloud Monitoring publish + // lag at the sample instant is not health evidence, and single-shot wave 0 + // failed a whole batch on a series that was fresh again a minute later. + assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" --retry-freshness/) + assert.doesNotMatch(job, /RETRY_ARGS/) assert.match(job, /timeout-minutes: 75/) // Both age gates step by the cell job timeout above; the constant is // duplicated across the two languages, so pin each copy to it. From 040c3e5b320a835d08875489673c74335040a312 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:31:25 -0400 Subject: [PATCH 04/26] fix(browser): match loading surfaces to the Orca theme (#18738) * fix(browser): theme unavailable guest surfaces without recoloring pages * test(browser): keep generated loading evidence out of the PR diff * test(browser): freeze recovery clock during artificial attach gate * test(browser): await painted content after network recovery --- .../BrowserPane.webview-preferences.test.ts | 2 +- .../browser-page-webview-surface.test.ts | 93 ++++++ .../host-guest/browser-page-webview.ts | 22 +- tests/e2e/browser-loading-surface-oracle.ts | 300 ++++++++++++++++++ tests/e2e/browser-loading-surface.spec.ts | 21 ++ 5 files changed, 433 insertions(+), 5 deletions(-) create mode 100644 src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts create mode 100644 tests/e2e/browser-loading-surface-oracle.ts create mode 100644 tests/e2e/browser-loading-surface.spec.ts diff --git a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts index 5e6dc77f67d..aff65e4eb3e 100644 --- a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts +++ b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts @@ -56,7 +56,7 @@ describe('BrowserPane webview preferences', () => { 'persist:orca-browser-session-profile-1' ) expect(ensuredWebview?.webview.getAttribute('webpreferences')).toBe( - ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE + `${ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE},transparent=false` ) expect(registryMocks.registerPersistentWebview).toHaveBeenCalledWith( 'browser-page-1', diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts new file mode 100644 index 00000000000..2d156ed926c --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts @@ -0,0 +1,93 @@ +// @vitest-environment happy-dom +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ORCA_BROWSER_BLANK_URL } from '../../../../../shared/constants' +import { ensureBrowserPageWebview } from './browser-page-webview' +import { webviewRegistry } from './webview-registry' + +vi.mock('./webview-registry', () => { + const webviewRegistry = new Map() + return { + webviewRegistry, + registerPersistentWebview: vi.fn((id, guest) => webviewRegistry.set(id, guest)), + replacePersistentWebview: vi.fn(), + destroyPersistentWebview: vi.fn() + } +}) + +afterEach(() => { + document.body.replaceChildren() + webviewRegistry.clear() +}) + +function createGuest(): Electron.WebviewTag { + const container = document.createElement('div') + document.body.appendChild(container) + return ensureBrowserPageWebview({ + browserTabId: 'surface-test', + container, + inputLocked: false, + webviewPartition: 'persist:browser-test', + resolveContainer: () => container + })!.webview +} + +function commit(guest: Electron.WebviewTag, url: string, isMainFrame = true): void { + guest.dispatchEvent(Object.assign(new Event('load-commit'), { url, isMainFrame })) +} + +describe('browser page surface ownership', () => { + it('themes the host before attach and uses an opaque native canvas for real pages', () => { + const guest = createGuest() + expect(guest.style.background).toBe('var(--background)') + expect(guest.getAttribute('webpreferences')).toContain('transparent=false') + expect(guest.getAttribute('webpreferences')).toContain('disableHtmlFullscreenWindowResize=true') + }) + + it.each(['about:blank', ORCA_BROWSER_BLANK_URL])( + 'keeps %s unavailable through first navigation, then reveals the committed page', + (url) => { + const guest = createGuest() + commit(guest, url) + expect(guest.style.visibility).toBe('hidden') + guest.dispatchEvent(new Event('did-start-loading')) + expect(guest.style.visibility).toBe('hidden') + commit(guest, 'https://example.test') + expect(guest.style.visibility).toBe('visible') + guest.dispatchEvent(new Event('did-start-loading')) + expect(guest.style.visibility).toBe('visible') + commit(guest, 'about:blank', false) + expect(guest.style.visibility).toBe('visible') + } + ) + + it('preserves a reused guest and initializes the same surface after a container remount', () => { + const guest = createGuest() + commit(guest, 'https://example.test') + const container = guest.parentElement as HTMLDivElement + const reused = ensureBrowserPageWebview({ + browserTabId: 'surface-test', + container, + inputLocked: false, + webviewPartition: 'persist:browser-test', + resolveContainer: () => container + })! + expect(reused.created).toBe(false) + expect(reused.webview).toBe(guest) + expect(reused.webview.style.visibility).toBe('visible') + const replacement = createGuest() + expect(replacement).not.toBe(guest) + expect(replacement.style.background).toBe('var(--background)') + expect(replacement.getAttribute('webpreferences')).toContain('transparent=false') + commit(replacement, ORCA_BROWSER_BLANK_URL) + expect(replacement.style.visibility).toBe('hidden') + }) + + it('exposes the themed host after renderer loss until a recovered document commits', () => { + const guest = createGuest() + commit(guest, 'https://example.test') + guest.dispatchEvent(new Event('render-process-gone')) + expect(guest.style.visibility).toBe('hidden') + commit(guest, 'https://example.test') + expect(guest.style.visibility).toBe('visible') + }) +}) diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts index 30e1cc18442..3c959751051 100644 --- a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts @@ -1,3 +1,4 @@ +import { ORCA_BROWSER_BLANK_URL } from '../../../../../shared/constants' import { ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE } from '../../../../../shared/browser-guest-web-preferences' import { destroyPersistentWebview, @@ -59,16 +60,29 @@ export function ensureBrowserPageWebview({ webview.setAttribute('allowpopups', '') // Why: Electron spreads the webpreferences keys verbatim, so the shared // camelCase attribute must stay intact for fullscreen containment to work. - webview.setAttribute('webpreferences', ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE) + // Keep Chromium's normal page canvas opaque while the host underneath follows Orca's theme. + webview.setAttribute( + 'webpreferences', + `${ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE},transparent=false` + ) webview.style.display = 'flex' webview.style.flex = '1' webview.style.width = '100%' webview.style.height = '100%' webview.style.border = 'none' setBrowserPageWebviewInputLock(webview, inputLocked) - // Why: some pages never paint a background, and a white viewport matches - // normal browser behavior instead of leaking Orca chrome through the guest. - webview.style.background = '#ffffff' + webview.style.background = 'var(--background)' + const guest = webview + // A committed synthetic blank document belongs to New Tab, including while its first URL waits. + guest.addEventListener('load-commit', (event) => { + if (event.isMainFrame) { + guest.style.visibility = + event.url === 'about:blank' || event.url === ORCA_BROWSER_BLANK_URL ? 'hidden' : 'visible' + } + }) + guest.addEventListener('render-process-gone', () => { + guest.style.visibility = 'hidden' + }) registerPersistentWebview(browserTabId, webview) activeContainer.appendChild(webview) created = true diff --git a/tests/e2e/browser-loading-surface-oracle.ts b/tests/e2e/browser-loading-surface-oracle.ts new file mode 100644 index 00000000000..5fab1c78c94 --- /dev/null +++ b/tests/e2e/browser-loading-surface-oracle.ts @@ -0,0 +1,300 @@ +import { createServer } from 'node:http' +import { writeFile } from 'node:fs/promises' +import type { AddressInfo } from 'node:net' +import { expect, type Page } from '@stablyai/playwright-test' +import { PNG } from 'pngjs' + +// Hold the response, not a timer: every screenshot precedes the first document commit. +export async function observeBrowserLoadingSurface( + page: Page, + outputPath: (name: string) => string, + crashGuest?: (id: number) => Promise +) { + const pendingResponses: (() => void)[] = [] + const release = (): void => { + pendingResponses.splice(0).forEach((send) => send()) + } + let requestCount = 0 + let flushPrefix: (() => void) | undefined + let retryReady = false + const server = createServer((request, response) => { + if (request.url === '/fail' && !retryReady) { + response.destroy() + return + } + if (request.url !== '/held' && request.url !== '/fail') { + response.writeHead(204).end() + return + } + requestCount += 1 + let prefixSent = false + flushPrefix = () => { + prefixSent = true + response.writeHead(200, { 'Content-Type': 'text/html' }) + response.write( + `Surface oracle` + ) + } + pendingResponses.push(() => + response.end( + `${prefixSent ? '' : 'Surface oracle'}

Usable webpage

` + ) + ) + }) + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const url = `http://127.0.0.1:${(server.address() as AddressInfo).port}/held` + const observations: Record[] = [] + // Freezing attachment for a screenshot must also freeze the missing-guest watchdog. + await page.clock.pauseAt(new Date()) + const attachGate = await page.evaluateHandle((heldUrl) => { + const original = Element.prototype.setAttribute + const pending: (() => void)[] = [] + Element.prototype.setAttribute = function (name, value) { + if (this.tagName === 'WEBVIEW' && name === 'src' && value === heldUrl) { + pending.push(() => original.call(this, name, value)) + return + } + original.call(this, name, value) + } + return () => { + Element.prototype.setAttribute = original + pending.splice(0).forEach((release) => release()) + } + }, url) + try { + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + await expect(page.locator('html')).toHaveClass(/dark/) + const tab = await page.evaluate((url) => { + const s = window.__store!.getState() + return s.createBrowserTab(s.activeWorktreeId!, url, { + activate: true, + title: 'Surface oracle' + }) + }, url) + let guest = page.locator(`[data-browser-overlay-tab-id="${tab.id}"] webview`) + await expect(guest).toHaveCount(1) + const capture = async (phase: string, expected: 'theme' | 'white', loading = false) => { + const state = await guest.evaluate((element) => { + const webview = element as Electron.WebviewTag + const rect = webview.closest('[data-browser-page-container]')!.getBoundingClientRect() + let loading = false + let url = '' + let attached = false + try { + loading = webview.isLoading() + url = webview.getURL() + attached = webview.getWebContentsId() > 0 + } catch { + /* Guest creation is held by the oracle. */ + } + return { + attached, + background: getComputedStyle(webview).backgroundColor, + theme: getComputedStyle(webview.closest('[data-browser-page-container]')!) + .backgroundColor, + visibility: getComputedStyle(webview).visibility, + display: getComputedStyle(webview).display, + loading, + url, + rect: { x: rect.x, y: rect.y, width: rect.width, height: rect.height } + } + }) + const target = + expected === 'white' ? [255, 255, 255] : state.theme.match(/\d+/g)!.slice(0, 3).map(Number) + let samples: number[][] = [] + const sampleSurface = async (): Promise => { + const screenshot = await page.screenshot({ path: outputPath(`${phase}.png`), scale: 'css' }) + const png = PNG.sync.read(screenshot) + samples = [0.08, 0.92].flatMap((x) => + [0.08, 0.85, 0.92].map((y) => { + const offset = + (Math.floor(state.rect.y + state.rect.height * y) * png.width + + Math.floor(state.rect.x + state.rect.width * x)) * + 4 + return [...png.data.subarray(offset, offset + 3)] + }) + ) + return samples.every((rgb) => rgb.every((c, i) => Math.abs(c - target[i]) <= 2)) + } + let pixelPass = await sampleSurface() + if (expected === 'white' && !pixelPass) { + // Loading can stop before the compositor presents the recovered guest's first frame. + await expect.poll(async () => (pixelPass = await sampleSurface())).toBe(true) + } + const statePass = + expected === 'theme' + ? state.background === state.theme || + state.visibility === 'hidden' || + state.display === 'none' + : state.url.startsWith('http://127.0.0.1:') + expect(state.loading).toBe(loading) + observations.push({ + phase, + ...state, + samples, + pixelPass, + statePass, + pass: pixelPass && statePass + }) + } + const dismissDrawHint = page.getByRole('button', { name: 'Got it', exact: true }) + if (await dismissDrawHint.isVisible()) { + await dismissDrawHint.click() + await expect(dismissDrawHint).not.toBeVisible() + } + await page.keyboard.press('Escape') + await page.mouse.move(0, 0) + await capture('pre-attach', 'theme') + await attachGate.evaluate((release) => release()) + await page.clock.resume() + await expect.poll(() => requestCount).toBe(1) + await capture('dark-held', 'theme', true) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'light' }) + }) + await expect(page.locator('html')).toHaveClass(/light/) + await capture('light-held', 'theme', true) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + await expect(page.locator('html')).toHaveClass(/dark/) + await capture('dark-again-held', 'theme', true) + await page.emulateMedia({ colorScheme: 'light' }) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'system' }) + }) + await expect(page.locator('html')).toHaveClass(/light/) + await capture('system-light-held', 'theme', true) + await page.emulateMedia({ colorScheme: 'dark' }) + await expect(page.locator('html')).toHaveClass(/dark/) + await capture('system-dark-held', 'theme', true) + await page.emulateMedia({ colorScheme: null }) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + flushPrefix!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).getTitle())) + .toBe('Surface oracle') + await capture('committed-empty', 'theme', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + expect( + await guest.evaluate((e) => + (e as Electron.WebviewTag).executeJavaScript( + '({ text: document.querySelector("h1").textContent, style: document.body.getAttribute("style") })' + ) + ) + ).toEqual({ text: 'Usable webpage', style: null }) + await capture('unstyled-painted', 'white') + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'light' }) + }) + await expect(page.locator('html')).toHaveClass(/light/) + await capture('unstyled-light', 'white') + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + await expect(page.locator('html')).toHaveClass(/dark/) + const pane = page.locator(`[data-browser-overlay-tab-id="${tab.id}"]`) + await pane.getByRole('button', { name: 'Reload', exact: true }).click() + await expect.poll(() => requestCount).toBe(2) + await capture('reload-retained', 'white', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + const guestId = await guest.evaluate((e) => (e as Electron.WebviewTag).getWebContentsId()) + const worktreeId = await page.evaluate(() => window.__store!.getState().activeWorktreeId!) + await page.evaluate(() => window.__store!.getState().setActiveWorktree(null)) + await expect(pane).not.toBeVisible() + await page.evaluate( + ({ worktreeId, tabId }) => { + const s = window.__store!.getState() + s.setActiveWorktree(worktreeId) + s.setActiveBrowserTab(tabId) + }, + { worktreeId, tabId: tab.id } + ) + await expect(pane).toBeVisible() + expect(await guest.evaluate((e) => (e as Electron.WebviewTag).getWebContentsId())).toBe(guestId) + await capture('unpark-retained', 'white') + if (crashGuest) { + await crashGuest(guestId) + await expect.poll(() => requestCount).toBe(3) + await capture('recovery-held', 'theme', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('recovery-painted', 'white') + } + const blank = await page.evaluate(() => { + const s = window.__store!.getState() + return s.createBrowserTab(s.activeWorktreeId!, 'about:blank', { activate: true }) + }) + guest = page.locator(`[data-browser-overlay-tab-id="${blank.id}"] webview`) + await expect + .poll(() => + guest.evaluate((e) => { + try { + return (e as Electron.WebviewTag).getURL() + } catch { + return '' + } + }) + ) + .toMatch(/about:blank|data:text\/html,/) + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await page.keyboard.press('Escape') + await capture('new-tab', 'theme') + // Navigate through the real address input, retaining the blank document while the server waits. + const address = page.locator(`[data-browser-overlay-tab-id="${blank.id}"] input`).first() + await address.fill(url) + await address.press('Enter') + await expect.poll(() => requestCount).toBe(crashGuest ? 4 : 3) + await capture('new-tab-first-navigation-held', 'theme', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('new-tab-painted', 'white') + await address.fill(url.replace('/held', '/fail')) + await address.press('Enter') + const retry = page + .locator(`[data-browser-overlay-tab-id="${blank.id}"]`) + .getByRole('button', { name: 'Retry', exact: true }) + .filter({ hasText: 'Retry' }) + await expect(retry).toBeVisible() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('network-error', 'theme') + retryReady = true + const countBeforeRetry = requestCount + await retry.click() + await expect.poll(() => requestCount).toBeGreaterThan(countBeforeRetry) + await capture('network-retry-held', 'theme', true) + release!() + await expect(retry).not.toBeVisible() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('network-retry-painted', 'white') + await writeFile(outputPath('observations.json'), JSON.stringify(observations, null, 2)) + return observations + } finally { + await attachGate.evaluate((release) => release()).catch(() => {}) + await page.clock.resume().catch(() => {}) + await attachGate.dispose() + release?.() + server.closeAllConnections() + await new Promise((resolve) => server.close(() => resolve())) + } +} diff --git a/tests/e2e/browser-loading-surface.spec.ts b/tests/e2e/browser-loading-surface.spec.ts new file mode 100644 index 00000000000..ac1f21eac57 --- /dev/null +++ b/tests/e2e/browser-loading-surface.spec.ts @@ -0,0 +1,21 @@ +import { expect, test } from './helpers/orca-app' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { crashGuestRenderer } from './browser-guest-runtime-oracle' +import { observeBrowserLoadingSurface } from './browser-loading-surface-oracle' + +test('browser host follows the theme before content and preserves the webpage canvas', async ({ + orcaPage, + electronApp +}, testInfo) => { + await waitForSessionReady(orcaPage) + await ensureTerminalVisible(orcaPage) + await waitForActiveWorktree(orcaPage) + const observations = await observeBrowserLoadingSurface( + orcaPage, + (name) => testInfo.outputPath(name), + async (id) => { + await crashGuestRenderer(electronApp, id) + } + ) + expect(observations.filter((entry) => !entry.pass)).toEqual([]) +}) From 436ef827dda5941940a072b754ee3162aadcee1b Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:47:15 -0400 Subject: [PATCH 05/26] fix(browser): present Electron's own user agent so Cloudflare Turnstile clears (#18749) Orca rewrote every browser session's UA to look like plain Chrome by stripping the Electron and app tokens. That rewrite is what Cloudflare rejects: a Chrome UA that ships no client hints reads as a spoof and Turnstile returns 600010, while the same binary on the same IP clears every challenge with its stock UA. PR #885 added the rewrite to fix 600010 and was treating a symptom it created; issue #11518 later found the same rewrite is what broke Google sign-in. - Keep the stock Electron UA on every partition. The webRequest handler now only owns the host-scoped Google auth Firefox switch, which stays unchanged. - Delete the anti-detection script. Measured on Electron 43: plugins are already a real PluginArray, window.chrome exists, and navigator.webdriver is false even with the debugger attached, so three of its four premises were wrong, and the overrides it installed (instance-level webdriver, non-native Permissions.query, stubbed chrome.csi/loadTimes) are themselves published bot signatures. - Stop attaching a CDP debugger to every browsing guest. Only the auth-UA detach listener remains, because a detach clears Chromium's standing UA override. - Stop sending Runtime.enable into cross-origin iframes when the agent bridge auto-attaches. The challenge widget is one, nothing reads iframe Runtime events, and the Runtime domain's serialization side effect is the documented Cloudflare CDP tell. - Add a real-Electron test proving the wire identity: stock UA to ordinary hosts, Firefox with no client hints to accounts.google.com. Verified in the dev build: dash.cloudflare.com/login no longer shows "There was a problem with verification" and scrapingcourse.com's managed challenge clears, both failing deterministically before. Fixes #13822 --- config/reliability-gates.jsonc | 2 +- docs/site/content/docs/browser/profiles.mdx | 4 +- .../anti-detection-permission-status.test.ts | 210 ------------------ src/main/browser/anti-detection.test.ts | 157 ------------- src/main/browser/anti-detection.ts | 161 -------------- src/main/browser/browser-google-auth-ua.ts | 8 +- .../browser-manager-auth-user-agent.test.ts | 9 +- .../browser-manager-guest-lifecycle.test.ts | 21 +- ...owser-manager-guest-policy-profile.test.ts | 14 +- .../browser/browser-manager-guest-policy.ts | 7 +- .../browser/browser-manager-navigation.ts | 22 +- src/main/browser/browser-manager-state.ts | 48 +--- src/main/browser/browser-manager-types.ts | 2 +- .../browser-manager-viewport-override.test.ts | 26 +-- .../browser-manager-viewport-test-fixtures.ts | 6 +- src/main/browser/browser-manager-viewport.ts | 2 +- ...browser-session-partition-policies.test.ts | 3 +- .../browser-session-partition-policies.ts | 14 +- ...er-session-partition-proxy-install.test.ts | 3 +- ...owser-session-registry.persistence.test.ts | 48 ++-- .../browser/browser-session-registry.test.ts | 182 +++------------ ...-session-ua-wire-identity.electron.test.ts | 180 +++++++++++++++ src/main/browser/browser-session-ua.ts | 87 +------- .../browser/browser-viewport-user-agent.ts | 4 +- .../browser-webauthn-profile-delete.test.ts | 3 +- src/main/browser/cdp-debugger-channel.ts | 7 - src/main/browser/cdp-debugger-events.ts | 4 +- src/main/browser/cdp-debugger-lifecycle.ts | 6 - .../browser/cdp-ws-proxy-focus-replay.test.ts | 21 +- src/main/browser/cdp-ws-proxy.test.ts | 18 +- .../window/main-window-webview-security.ts | 2 +- tests/tools/google-signin-ua-probe.cjs | 6 +- 32 files changed, 335 insertions(+), 952 deletions(-) delete mode 100644 src/main/browser/anti-detection-permission-status.test.ts delete mode 100644 src/main/browser/anti-detection.test.ts delete mode 100644 src/main/browser/anti-detection.ts create mode 100644 src/main/browser/browser-session-ua-wire-identity.electron.test.ts diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index b7905fa0419..9f5885edc8a 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -2855,7 +2855,7 @@ "https://github.com/stablyai/orca/pull/13876" ], "invariant": "Opening one HTML preview from a paired client renders the workspace document in exactly one client-local browser tab, located by that document and served over the orca-preview scheme. The client gains exactly that one browser workspace and it is the document one — blank where a URL page carries a URL, named by the document, with the chip naming the file — while the host gains no browser page at all, neither in its own page registry nor in the tab snapshot its clients publish into. The preview occupies its own split without taking focus from the source editor; an explicit click activates it, and closing it removes only the preview. Following a document from a file link is the other half of that switch and does move the reader to it, tab group included, whether the preview is new or already open, because opening a file is a request to look at it. A document tab quit with the client comes back as the same row on a grant the relaunched client mints afresh. A preview is named by the browser page it is open in, not by a namespace of its own, and the page registry has two halves: a workspace-document guest is registered in its own map and is absent from the browsing one entirely. That absence is the fence. Page, session and profile management, agent tab enumeration and command targeting, download routing and certificate attribution all read the browsing map directly, in more places than a per-channel guard could be remembered in, so none of them can name a document page and none of them carries a guard. Browser tools the reader drives (element grab, hover describe, selection capture, the annotation viewport bridge) are the one operation that legitimately spans the halves, and they go through the single authority that reads both, keyed by the page and its hosting renderer. The halves are disjoint in both directions: browsing registration refuses a page the document half already holds, and minting a grant refuses a page the browsing half already holds, so one id can never name a surface in both. The headless backend acts on that refusal by destroying the window it had already opened rather than leaving a policy-less page behind an id nothing can drive, keeping nothing under that id for its own shutdown to hand back. Registration refuses on the same terms when the guest it was asked about is already gone. The exit door is guarded in both its halves: a preview withdraws by revoking its grant and never through the unregister channel, so a page the document half holds arriving there is refused before either the registration teardown or the grab-state disposal beside it, which would otherwise drop the intent an in-flight preview grab compares by identity and leave that grab answering ok without ever arming its guest. A bridge request whose guest does not resolve is refused without tearing down the page it named, so a misaddressed request cannot cancel a healthy page's in-flight downloads and grabs. The annotation viewport bridge resolves its guest when its serialized op actually runs rather than when the request arrived, so a cross-process navigation while it waited cannot leave the bridge installed in a retired guest while the reader looks at a new one. State main keys by a preview's page is disposed when that page's grant is revoked, which is the only signal a preview's surface is gone. A tool asking for a page whose guest has not attached yet waits for that registration and arms when it arrives, rather than answering not-ready at the reader; that wait resolves only the request already naming this page, never the worktree-wide or any-tab waits the CLI and agents use to ask for a browser tab to drive. Handing the previewed document to the reader's own machine routes on the owners its grant was minted against — the file's own connection owner and the worktree's own runtime owner, neither read from the tab's stored fields. Only a document proven to live on this machine reaches the client OS; one with a resolved remote owner is downloaded first; and one whose owner cannot be resolved at all, workspace root included, is refused with a message naming that, because the download route would otherwise read the same absolute path on the client and hand back a same-named local file under the remote document's name. A runtime-owned path that falls outside its worktree root is refused by that route itself and surfaces as a failure toast rather than a download. Nothing the document does writes a file to this machine either: the preview partition denies downloads outright instead of routing them through the browser download flow, which has no page to attribute a preview's bytes to and would otherwise reserve a name in this desktop's Downloads folder and write them there unprompted. That refusal is visible to the reader and invisible to the document: the preview's shell carries a fixed sentence saying downloads are off, published at most once per preview per interval so a document asking in a loop cannot fill Orca's chrome, while the page itself gets back exactly what it got before, which is nothing. The sentence names no file, because the document chooses the name it offers; and a refusal never takes the document away the way an entry document's own failure does, whatever it names. A preview is a browser tab, not an editor tab in a preview mode: it is named the way a browser tab is named — by the document it shows when that document declares a title, and by the file it shows when it does not — while the chip goes on naming the file and the host whatever the document calls itself. A title is refused on the same terms the url is: a document that declares none has Chromium report the grant URL as its title, and that title is stored, mirrored onto the tab and written to disk, so anything carrying the scheme falls back to the file instead. It is created by the preview action as a page located by its document, it carries the workspace-relative path copy the editor's path header owned, and closing it revokes the grant that made the document readable while a URL tab closing beside it revokes nothing. Chrome persisted by builds that made previews editor tabs is dropped on restore rather than coming back naming a surface no restore can produce, and the ordinary editor tab for the same document is left alone. A document tab is held back at the mobile publish boundary — no client holds its grant, and the wire has no tab kind for it — while an ordinary browser tab beside it still publishes. It is held back from the group projection that publishes tab order, recency and group activity as well as from the tab list itself, so no published group names a tab the phone is never sent. A browser page can be located by a workspace document instead of a URL, and the document is the whole of its stored identity. The grant and the orca-preview URL that document is served over are minted when the page mounts and replaced by a hard reload, so neither is ever written to the page's url, mirrored onto its tab, persisted or published: such a page's url is the blank URL from creation through restore, including when a session written elsewhere carries a grant URL in, and what the session carries is the worktree and path a restored page mints afresh against today's owners. Every door onto a page's url holds that line — creation, the title update, and the navigation commit alike — so a report about a document page cannot give it a URL it never had, and the title fallback and the loading affordance follow the url each door actually wrote. The mirror carries the document too, so a tab entry cannot go on naming a document its active page has left. Every guest in the app is policy-attached through one door: a workspace document takes a restricted profile there rather than a separate installer beside it, so the attachment bookkeeping that door owns — what registration refuses, and what teardown frees — covers a preview on the same terms as a browsing page, and a preview takes none of the browsing machinery that door installs. That authority answers from the moment the embedder hands the guest over rather than only after a later navigation: the guest binds to the grant it is already showing, so the tools reach the document the reader opened and not just one they navigated to. A read the host reports as truncated or over-cap is refused rather than served partially, and a document outside the paired worktree is refused with a message naming that boundary instead of a bare read failure. The rendered document reaches nothing off-machine on its own: every served response carries a self-only content security policy, the preview session cancels any request that is not in-document, subframes cannot navigate outside the grant, a guest no document has yet bound to a grant may not navigate at all, the guest gathers no ICE candidates, and an SSH path that canonicalizes outside the grant root is refused before it is read. The one route out is a link the reader presses: a trusted click on an anchor, reported by the preview's own preload from a guest still bound to a live grant, leaves as an Orca browser tab rather than a native window or a dead click — and only after the reader confirms the exact destination URL, so a document cannot spend a single stray press exfiltrating what it can read into a link it authored. The preview hands its guest that focus itself whenever it is the surface the reader is in — a browsing page gets it from the chrome around it, and a preview has no chrome to get it from — and it does so only then, so a preview mounted behind a terminal or an editor never takes the keyboard from what the reader is actually in. It offers again when the window itself takes focus back and nothing in the embedder has claimed that focus, because another app coming to the front lands focus on the embedder rather than the guest and the route out would otherwise stay shut until something remounted the pane — while the same window focus also arrives when the reader presses a tab, that being the guest's own blur returning, and taking focus back from there would fight the reader for their own click. Nothing else does. A navigation or popup the document starts by itself is swallowed whatever else is happening, including immediately after a genuine press elsewhere in the document, so a page that can read its grant cannot hand it to a browser tab; a middle click opens nothing; and a fragment link is answered inside the document. A preview attach carries the preview preload and no renderer-supplied one, and no other attach path can acquire it. A subresource the workspace will not send degrades the document to a notice naming that file, never to a failure panel over a page that rendered. A grant outlives neither the tab that owns it nor the renderer document that minted it, and only the trusted renderer can mint or revoke one. For the browser creations this gate still owns, owner-pinned creation returns the canonical host page identity before navigation readiness; delayed navigation cannot turn a created page into an unidentifiable failure or a duplicate retry. Capability rejection before host mutation must preserve the original error, issue no RPC, surface a failure toast, and remove only a caller-declared newly-created empty split. Post-create reconciliation failure requires exact rollback; ambiguous rollback rejects without local fallback.", - "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and anti-detection while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", + "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and auth-identity detach tracking while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-browser.test.ts src/main/runtime/rpc/methods/browser.test.ts src/renderer/src/lib/file-preview.test.ts src/renderer/src/runtime/web-session-browser-placement.test.ts src/renderer/src/runtime/web-runtime-session.test.ts src/renderer/src/runtime/web-session-tabs-sync.test.ts src/renderer/src/runtime/remote-server-parity.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/runtime/web-runtime-browser-materialization.test.ts", diff --git a/docs/site/content/docs/browser/profiles.mdx b/docs/site/content/docs/browser/profiles.mdx index 97286a6cb0c..102dd1e443d 100644 --- a/docs/site/content/docs/browser/profiles.mdx +++ b/docs/site/content/docs/browser/profiles.mdx @@ -9,9 +9,7 @@ Browser-use profiles let you run the Orca browser with a specific identity — a 1. Open [Settings → Browser → Profiles](/docs/settings). 1. Click **Add profile**, give it a name. 1. Optionally seed it with cookies, a user-agent, and a viewport size. -1. For sites that reject Orca's default Chrome-shaped UA (some Google sign-in flows), create a profile that keeps the **native Electron user agent** instead of spoofing. Default profiles still use the cleaned Chrome UA for broader Cloudflare compatibility. - -You can also create a no-spoof profile from the CLI with `orca tab profile create --no-ua-spoof` when you script browser setup. +1. Every profile presents Electron's own user agent. Orca no longer rewrites it to look like Chrome, because Cloudflare Turnstile rejects a Chrome-shaped UA that sends no client hints and accepts a declared Electron client. The only exception is Google's sign-in hosts, where Orca presents a Firefox identity so Google issues cookies bound to the embedded browser. A **native user agent** profile (`orca tab profile create --no-ua-spoof`) also skips that Google exception. ## Cookie import and Google sign-in diff --git a/src/main/browser/anti-detection-permission-status.test.ts b/src/main/browser/anti-detection-permission-status.test.ts deleted file mode 100644 index d686e32c975..00000000000 --- a/src/main/browser/anti-detection-permission-status.test.ts +++ /dev/null @@ -1,210 +0,0 @@ -import { runInNewContext } from 'node:vm' -import { describe, expect, it } from 'vitest' - -import { ANTI_DETECTION_SCRIPT } from './anti-detection' - -type PermissionQueryResult = EventTarget & { - state: string - onchange: EventListener | null - marker: string -} - -type PermissionStatusConstructor = { - new (): PermissionQueryResult - prototype: PermissionQueryResult -} - -type AntiDetectionContext = { - Notification: { - permission: string - requestPermission: (callback?: (permission: string) => void) => Promise - } - PermissionStatus: PermissionStatusConstructor - dispatchPermissionChange: (name: string) => void - navigator: { - permissions: { - query: (descriptor: { name: string }) => Promise - } - } -} - -function createContext(args: { - nativeNotificationPermission: string - requestedNotificationPermission: string - rejectedPermissions?: string[] -}): AntiDetectionContext & Record { - class PermissionStatus extends EventTarget { - #state = 'denied' - #onchange: EventListener | null = null - marker = 'real-status' - - get state(): string { - return this.#state - } - - get onchange(): EventListener | null { - return this.#onchange - } - - set onchange(listener: EventListener | null) { - if (this.#onchange) { - super.removeEventListener('change', this.#onchange) - } - this.#onchange = typeof listener === 'function' ? listener : null - if (this.#onchange) { - super.addEventListener('change', this.#onchange) - } - } - } - - const statuses = new Map() - const rejectedPermissions = new Set(args.rejectedPermissions) - - class Permissions { - query(descriptor: { name: string }): Promise { - if (rejectedPermissions.has(descriptor.name)) { - return Promise.reject(new Error('Unsupported permission')) - } - const status = new PermissionStatus() - const permissionStatuses = statuses.get(descriptor.name) ?? [] - permissionStatuses.push(status) - statuses.set(descriptor.name, permissionStatuses) - return Promise.resolve(status) - } - } - - const Notification = { - permission: args.nativeNotificationPermission, - requestPermission(callback?: (permission: string) => void): Promise { - callback?.(args.requestedNotificationPermission) - return Promise.resolve(args.requestedNotificationPermission) - } - } - Object.defineProperty(Notification, 'permission', { - configurable: true, - get: () => args.nativeNotificationPermission - }) - - return { - Date, - Event, - EventTarget, - Object, - Promise, - Set, - performance: { now: () => 0 }, - // Why: the script's Firefox gate reads navigator.userAgent, so a non-Firefox UA keeps these - // tests on the ordinary-page path where the PermissionStatus override applies. - window: { chrome: {} }, - navigator: { - userAgent: - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/150.0.0.0 Safari/537.36', - plugins: [], - languages: [], - permissions: new Permissions() - }, - Permissions, - PermissionStatus, - Notification, - dispatchPermissionChange(name: string): void { - for (const status of statuses.get(name) ?? []) { - status.dispatchEvent(new Event('change')) - } - } - } as AntiDetectionContext & Record -} - -describe('ANTI_DETECTION_SCRIPT — PermissionStatus', () => { - it('keeps an existing notification status current after permission changes', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'notifications' }) - - expect(context.Notification.permission).toBe('default') - expect(status.state).toBe('prompt') - - await expect(context.Notification.requestPermission()).resolves.toBe('granted') - - expect(context.Notification.permission).toBe('granted') - expect(status.state).toBe('granted') - }) - - it('preserves native PermissionStatus identity and methods', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'camera' }) - const expectedSource = Function.prototype.toString.call( - context.PermissionStatus.prototype.addEventListener - ) - - expect(status).toBeInstanceOf(context.PermissionStatus) - expect(status.state).toBe('prompt') - expect(status.constructor.name).toBe('PermissionStatus') - expect(status.marker).toBe('real-status') - expect(status.addEventListener.name).toBe('addEventListener') - expect(status.addEventListener).toBe(status.addEventListener) - expect(Function.prototype.toString.call(status.addEventListener)).toBe(expectedSource) - expect(expectedSource).toContain('addEventListener') - }) - - it('delivers change events through the returned status with the overridden state', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'notifications' }) - const events: { receiver: EventTarget; target: EventTarget | null; state: string }[] = [] - const recordEvent = function (this: EventTarget, event: Event): void { - events.push({ - receiver: this, - target: event.target, - state: (event.target as PermissionQueryResult).state - }) - } - - status.addEventListener('change', recordEvent) - expect(() => { - status.onchange = function (this: EventTarget, event): void { - recordEvent.call(this, event) - } - }).not.toThrow() - - await context.Notification.requestPermission() - context.dispatchPermissionChange('notifications') - - expect(events).toHaveLength(2) - expect(events).toEqual([ - { receiver: status, target: status, state: 'granted' }, - { receiver: status, target: status, state: 'granted' } - ]) - }) - - // Why: 'camera' rather than 'storage-access' — #14685 narrowed the intercepted set to - // camera/microphone, so a name outside it falls through to the real query and never reaches - // the fallback at all. - it('uses a non-enumerable EventTarget fallback when the native query rejects', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted', - rejectedPermissions: ['camera'] - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'camera' }) - - expect(status).toBeInstanceOf(EventTarget) - expect(status).not.toBeInstanceOf(context.PermissionStatus) - expect(status.state).toBe('prompt') - expect(Object.keys(status)).toEqual([]) - }) -}) diff --git a/src/main/browser/anti-detection.test.ts b/src/main/browser/anti-detection.test.ts deleted file mode 100644 index ddb904cb436..00000000000 --- a/src/main/browser/anti-detection.test.ts +++ /dev/null @@ -1,157 +0,0 @@ -import { runInNewContext } from 'node:vm' -import { describe, expect, it } from 'vitest' - -import { ANTI_DETECTION_SCRIPT } from './anti-detection' -import { googleAuthUserAgent } from './browser-google-auth-ua' - -type PermissionQueryResult = { - state: string - onchange: null -} - -type AntiDetectionContext = { - Notification: { - permission: string - requestPermission: (callback?: (permission: string) => void) => Promise - } - navigator: { - userAgent: string - permissions: { - query: (descriptor: { name: string }) => Promise - } - } - window: { - chrome?: { - runtime?: unknown - csi?: () => unknown - loadTimes?: () => unknown - } - } -} - -function createContext(args: { - nativeNotificationPermission: string - requestedNotificationPermission: string - userAgent?: string -}): AntiDetectionContext & Record { - class Permissions { - query(): Promise { - return Promise.resolve({ state: 'denied', onchange: null }) - } - } - - const Notification = { - permission: args.nativeNotificationPermission, - requestPermission(callback?: (permission: string) => void): Promise { - callback?.(args.requestedNotificationPermission) - return Promise.resolve(args.requestedNotificationPermission) - } - } - Object.defineProperty(Notification, 'permission', { - configurable: true, - get: () => args.nativeNotificationPermission - }) - - return { - Date, - Object, - Promise, - Set, - performance: { now: () => 0 }, - // Electron 43 exposes this native object before the anti-detection script runs. - window: { chrome: {} }, - navigator: { - userAgent: - args.userAgent ?? - 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko) Chrome/151.0.0.0 Safari/537.36', - plugins: [], - languages: [], - permissions: new Permissions() - }, - Permissions, - Notification - } as AntiDetectionContext & Record -} - -describe('ANTI_DETECTION_SCRIPT', () => { - it('does not expose Chrome globals under a Firefox identity', () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied', - userAgent: googleAuthUserAgent() - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.window.chrome).toBeUndefined() - expect('chrome' in context.window).toBe(false) - }) - - it('keeps Chrome API stubs aligned with an ordinary Chrome page', () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.window.chrome?.runtime).toBeUndefined() - expect(context.window.chrome?.csi).toBeTypeOf('function') - expect(context.window.chrome?.loadTimes).toBeTypeOf('function') - }) - - it.each(['geolocation', 'idle-detection', 'midi', 'storage-access'])( - 'passes non-intercepted permission queries through to the native state for %s', - async (name) => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - await expect(context.navigator.permissions.query({ name })).resolves.toEqual({ - state: 'denied', - onchange: null - }) - } - ) - - it('reports notification permission as granted after a site permission request succeeds', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.Notification.permission).toBe('default') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'prompt', - onchange: null - }) - - await expect(context.Notification.requestPermission()).resolves.toBe('granted') - - expect(context.Notification.permission).toBe('granted') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'granted', - onchange: null - }) - }) - - it('preserves notification permission when Electron already reports a grant', async () => { - const context = createContext({ - nativeNotificationPermission: 'granted', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.Notification.permission).toBe('granted') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'granted', - onchange: null - }) - }) -}) diff --git a/src/main/browser/anti-detection.ts b/src/main/browser/anti-detection.ts deleted file mode 100644 index d725fbf965f..00000000000 --- a/src/main/browser/anti-detection.ts +++ /dev/null @@ -1,161 +0,0 @@ -// Why: Cloudflare Turnstile and similar bot detectors probe multiple browser -// APIs beyond navigator.webdriver. This script runs via -// Page.addScriptToEvaluateOnNewDocument before any page JS to mask automation -// signals that CDP debugger attachment and Electron's webview expose. -export const ANTI_DETECTION_SCRIPT = `(function() { - Object.defineProperty(navigator, 'webdriver', { get: () => false }); - // Why: Electron webviews expose an empty plugins array. Real Chrome always - // has at least a few default plugins (PDF Viewer, etc.). An empty array is - // a strong automation signal. - if (navigator.plugins.length === 0) { - Object.defineProperty(navigator, 'plugins', { - get: () => [ - { name: 'Chrome PDF Plugin', filename: 'internal-pdf-viewer' }, - { name: 'Chrome PDF Viewer', filename: 'mhjfbmdgcfjbbpaeojofohoefgiehjai' }, - { name: 'Native Client', filename: 'internal-nacl-plugin' } - ] - }); - } - // Why: auth hosts present Firefox, where Electron's native window.chrome is an identity mismatch. - if (navigator.userAgent.includes('Firefox/')) { - try { - delete window.chrome; - if ('chrome' in window) { - window.chrome = undefined; - } - } catch {} - } else { - // Why: Electron webviews may not have the window.chrome object that real - // Chrome exposes. Turnstile checks for its presence. The csi() and - // loadTimes() stubs satisfy deeper probes of Chrome-specific APIs. - if (!window.chrome) { - window.chrome = {}; - } - if (!window.chrome.csi) { - window.chrome.csi = function() { - return { - startE: Date.now(), - onloadT: Date.now(), - pageT: performance.now(), - tran: 15 - }; - }; - } - if (!window.chrome.loadTimes) { - window.chrome.loadTimes = function() { - return { - commitLoadTime: Date.now() / 1000, - connectionInfo: 'h2', - finishDocumentLoadTime: Date.now() / 1000, - finishLoadTime: Date.now() / 1000, - firstPaintAfterLoadTime: 0, - firstPaintTime: Date.now() / 1000, - navigationType: 'Other', - npnNegotiatedProtocol: 'h2', - requestTime: Date.now() / 1000 - 0.16, - startLoadTime: Date.now() / 1000 - 0.3, - wasAlternateProtocolAvailable: false, - wasFetchedViaSpdy: true, - wasNpnNegotiated: true - }; - }; - } - } - // Why: Electron's Permission API defaults to 'denied' for most permissions, - // but real Chrome returns 'prompt' for ungranted permissions. Returning - // 'denied' is a strong bot signal. Override the query result for common - // permissions that Turnstile and similar detectors probe. - var notificationPermission = 'default'; - var setNotificationPermission = function(permission) { - if (permission === 'granted' || permission === 'denied') { - notificationPermission = permission; - return permission; - } - notificationPermission = 'default'; - return 'default'; - }; - var notificationPermissionState = function() { - return notificationPermission === 'default' ? 'prompt' : notificationPermission; - }; - try { - if (Notification.permission === 'granted') { - notificationPermission = 'granted'; - } - } catch {} - const promptPerms = new Set([ - 'camera', 'microphone' - ]); - const origQuery = Permissions.prototype.query; - // Why: sites must receive the genuine PermissionStatus so native events, brand checks and method - // identity survive. Shadow only state, and resolve it lazily so existing statuses stay current. - function withOverriddenState(realStatus, stateProvider) { - Object.defineProperty(realStatus, 'state', { - configurable: true, - get: stateProvider - }); - return realStatus; - } - // Why: some names the real implementation rejects outright; fall back to an EventTarget so - // listener registration still works instead of throwing. - function fallbackStatus(stateProvider) { - const status = new EventTarget(); - Object.defineProperties(status, { - state: { configurable: true, get: stateProvider }, - onchange: { configurable: true, value: null, writable: true } - }); - return status; - } - function queryWithState(permissions, desc, stateProvider) { - let real; - try { - real = origQuery.call(permissions, desc); - } catch { - return Promise.resolve(fallbackStatus(stateProvider)); - } - return Promise.resolve(real).then( - (status) => withOverriddenState(status, stateProvider), - () => fallbackStatus(stateProvider) - ); - } - Permissions.prototype.query = function(desc) { - if (desc.name === 'notifications') { - return queryWithState(this, desc, notificationPermissionState); - } - if (promptPerms.has(desc.name)) { - return queryWithState(this, desc, () => 'prompt'); - } - return origQuery.call(this, desc); - }; - // Why: Electron may report Notification.permission as 'denied' by default - // whereas real Chrome reports 'default' for sites that haven't been granted - // or blocked. Turnstile cross-references this with the Permissions API. - try { - Object.defineProperty(Notification, 'permission', { - get: () => notificationPermission - }); - const origRequestPermission = Notification.requestPermission; - if (typeof origRequestPermission === 'function') { - Notification.requestPermission = function(callback) { - var wrappedCallback = typeof callback === 'function' - ? function(permission) { - callback(setNotificationPermission(permission)); - } - : undefined; - var result = origRequestPermission.call(Notification, wrappedCallback); - if (result && typeof result.then === 'function') { - return result.then(function(permission) { - return setNotificationPermission(permission); - }); - } - return result; - }; - } - } catch {} - // Why: Electron webviews may have an empty languages array. Real Chrome - // always has at least one entry. An empty array is an automation signal. - if (!navigator.languages || navigator.languages.length === 0) { - Object.defineProperty(navigator, 'languages', { - get: () => ['en-US', 'en'] - }); - } -})()` diff --git a/src/main/browser/browser-google-auth-ua.ts b/src/main/browser/browser-google-auth-ua.ts index 16ed14eb80f..e9b802f6d70 100644 --- a/src/main/browser/browser-google-auth-ua.ts +++ b/src/main/browser/browser-google-auth-ua.ts @@ -1,12 +1,12 @@ // Why: Google binds a signed-in session to the browser identity that created it. -// Cookies copied in from another browser (or sent under an Electron/Chrome-shaped -// UA that doesn't match a real first-party browser) get flagged by anti-fraud on -// accounts.google.com and expire within ~1h. Presenting a Firefox identity scoped +// Cookies copied in from another browser (or sent under a UA that doesn't match a +// real first-party browser) get flagged by anti-fraud on accounts.google.com and +// expire within ~1h. Presenting a Firefox identity scoped // to Google's auth hosts lets the user sign in *inside* the embedded browser, so // Google issues cookies bound to THIS browser that self-refresh — instead of us // transplanting cookies that go stale. Scope is deliberately the auth hosts only: // post-auth app surfaces (mail.google.com, myaccount.google.com, drive, etc.) keep -// the profile's real Chrome-shaped identity so nothing else about the session shifts. +// the profile's real identity so nothing else about the session shifts. // Why: exact hostname match — subdomains such as myaccount.google.com are post-auth // app surfaces, not the sign-in flow, and must retain the profile's real identity. diff --git a/src/main/browser/browser-manager-auth-user-agent.test.ts b/src/main/browser/browser-manager-auth-user-agent.test.ts index 405b689e850..eb0bd75b354 100644 --- a/src/main/browser/browser-manager-auth-user-agent.test.ts +++ b/src/main/browser/browser-manager-auth-user-agent.test.ts @@ -51,7 +51,7 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_CLEAN_UA + GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' const { @@ -197,8 +197,9 @@ describe('browserManager', () => { // Why: popup child windows get attachGuestPolicies but are never entered into tabIdByWebContentsId, // so a direct lookup of the UA mode misses the native opt-out. That is worse than doing nothing — - // native sessions skip setupClientHintsOverride, so the popup would send the raw Electron UA on the - // wire while navigator.userAgent claimed Firefox. Google sign-in popups are a first-class surface. + // native sessions never install the header-level Firefox switch, so the popup would send the + // Electron UA on the wire while navigator.userAgent claimed Firefox. Google sign-in popups are a + // first-class surface. it('leaves the UA untouched on auth hosts for a popup owned by a native-UA profile', () => { const ownerGuest = { id: 415, @@ -543,7 +544,7 @@ describe('browserManager', () => { ) expect(uaWrites.length).toBeGreaterThan(0) for (const [, params] of uaWrites) { - expect((params as { userAgent: string }).userAgent).toBe(GUEST_CLEAN_UA) + expect((params as { userAgent: string }).userAgent).toBe(GUEST_ELECTRON_UA) } }) }) diff --git a/src/main/browser/browser-manager-guest-lifecycle.test.ts b/src/main/browser/browser-manager-guest-lifecycle.test.ts index c8550083dfc..e136f532474 100644 --- a/src/main/browser/browser-manager-guest-lifecycle.test.ts +++ b/src/main/browser/browser-manager-guest-lifecycle.test.ts @@ -637,9 +637,9 @@ describe('browserManager', () => { ).toHaveLength(2) }) - it('cancels pending anti-detection reattach timers when unregistering a guest', () => { - vi.useFakeTimers() - + // Why: a plain browsing tab must never attach a debugger (Cloudflare treats CDP as a bot signal); + // the only debugger wiring it keeps is the detach listener that invalidates the auth-host UA override. + it('never attaches a debugger to a browsing guest and drops its detach listener on unregister', () => { const debuggerHandlers = new Map void>() const debuggerAttachMock = vi.fn() const guest = { @@ -670,18 +670,17 @@ describe('browserManager', () => { browserManager.attachGuestPolicies(guest as never) browserManager.registerGuest({ - browserPageId: 'browser-reattach', + browserPageId: 'browser-no-debugger', webContentsId: 809, rendererWebContentsId }) - debuggerHandlers.get('detach')?.() - expect(vi.getTimerCount()).toBe(1) + expect(debuggerAttachMock).not.toHaveBeenCalled() + expect(guest.debugger.sendCommand).not.toHaveBeenCalled() + expect(debuggerHandlers.has('detach')).toBe(true) - browserManager.unregisterGuest('browser-reattach') - expect(vi.getTimerCount()).toBe(0) - - vi.advanceTimersByTime(500) - expect(debuggerAttachMock).toHaveBeenCalledTimes(1) + browserManager.unregisterGuest('browser-no-debugger') + expect(debuggerHandlers.has('detach')).toBe(false) + expect(debuggerAttachMock).not.toHaveBeenCalled() }) }) diff --git a/src/main/browser/browser-manager-guest-policy-profile.test.ts b/src/main/browser/browser-manager-guest-policy-profile.test.ts index 47871f64ea9..c5c4c2521fe 100644 --- a/src/main/browser/browser-manager-guest-policy-profile.test.ts +++ b/src/main/browser/browser-manager-guest-policy-profile.test.ts @@ -52,6 +52,8 @@ type GuestFake = { isAttached: () => boolean attach: ReturnType sendCommand: ReturnType + on: ReturnType + off: ReturnType } on: (event: string, listener: (...args: never[]) => void) => void once: (event: string, listener: (...args: never[]) => void) => void @@ -81,7 +83,9 @@ function createGuest(id: number, url: string): GuestFake { debugger: { isAttached: () => true, attach: vi.fn(), - sendCommand: vi.fn(async () => undefined) + sendCommand: vi.fn(async () => undefined), + on: vi.fn(), + off: vi.fn() }, on: (event, listener) => { listeners.set(event, [...(listeners.get(event) ?? []), listener]) @@ -143,7 +147,7 @@ describe('guest policy profiles', () => { // The presence half of every absence below: a browsing guest observably takes all of it through // the same method, so a profile that fenced nothing — or an attach path that stopped installing // anything at all — cannot pass these by being uniformly empty. - it('gives a browsing guest link routing, popups and anti-detection', () => { + it('gives a browsing guest link routing, popups and auth-identity detach tracking', () => { const guest = createGuest(300, 'https://example.com/') browserManager.attachGuestPolicies(guest as never) @@ -151,7 +155,10 @@ describe('guest policy profiles', () => { expect(listenerCount(guest, 'dom-ready')).toBe(1) expect(listenerCount(guest, 'frame-created')).toBe(1) expect(listenerCount(guest, 'did-create-window')).toBe(1) - expect(guest.debugger.sendCommand).toHaveBeenCalled() + expect(guest.debugger.on).toHaveBeenCalledWith('detach', expect.any(Function)) + // Why: a plain browsing tab must never attach a debugger; Cloudflare treats CDP as a bot signal. + expect(guest.debugger.attach).not.toHaveBeenCalled() + expect(guest.debugger.sendCommand).not.toHaveBeenCalled() expect(navigateTo(guest, 'https://elsewhere.example/')).toBe(false) }) @@ -161,6 +168,7 @@ describe('guest policy profiles', () => { expect(listenerCount(guest, 'dom-ready')).toBe(0) expect(listenerCount(guest, 'frame-created')).toBe(0) expect(listenerCount(guest, 'did-create-window')).toBe(0) + expect(guest.debugger.on).not.toHaveBeenCalled() expect(guest.debugger.sendCommand).not.toHaveBeenCalled() expect(guest.executeJavaScriptInIsolatedWorld).not.toHaveBeenCalled() }) diff --git a/src/main/browser/browser-manager-guest-policy.ts b/src/main/browser/browser-manager-guest-policy.ts index c0d522235c8..3952a31c08b 100644 --- a/src/main/browser/browser-manager-guest-policy.ts +++ b/src/main/browser/browser-manager-guest-policy.ts @@ -35,8 +35,7 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean this.clickedLinkFrameNameByGuestId.set(guest.id, clickedLinkFrameName) } - // Why: bot detectors probe APIs that differ in Electron webviews; inject overrides each load so manual browsing passes. - const disposeAntiDetection = this.injectAntiDetection(guest) + const disposeAuthDetachTracking = this.trackDebuggerDetachForAuthUserAgent(guest) // Why: disable throttling so background screenshots still get frames; else the compositor stalls and capture returns empty. guest.setBackgroundThrottling(false) const disposePopupPolicy = this.installGuestPopupPolicy(guest, clickedLinkFrameName) @@ -44,14 +43,14 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean // Why: store cleanup so unregisterGuest can drop these listeners on teardown and let the WebContents wrapper GC. this.policyCleanupByGuestId.set(guest.id, () => { - disposeAntiDetection() + disposeAuthDetachTracking() disposePopupPolicy() disposeNavigationPolicy() }) } /** - * A workspace document is not the web: no popups, no link routing, no anti-detection, and no + * A workspace document is not the web: no popups, no link routing, no auth-identity tracking, and no * navigation bookkeeping for chrome it does not have. What it does share with a browsing guest is * this method's teardown, so a retired preview drops its listeners on the same path. */ diff --git a/src/main/browser/browser-manager-navigation.ts b/src/main/browser/browser-manager-navigation.ts index 4e061d288aa..5cb46ccb682 100644 --- a/src/main/browser/browser-manager-navigation.ts +++ b/src/main/browser/browser-manager-navigation.ts @@ -1,5 +1,4 @@ import { openPopupWithOriginBar, type PopupChildWindowOptions } from './popup-origin-bar-window' -import { cleanElectronUserAgent } from './browser-session-ua' import { getBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { googleAuthUserAgent, isGoogleAuthUrl } from './browser-google-auth-ua' import { buildViewportUserAgentOverride } from './browser-viewport-user-agent' @@ -12,10 +11,10 @@ import { BrowserManagerVisibility } from './browser-manager-visibility' export abstract class BrowserManagerNavigation extends BrowserManagerVisibility { // Why: navigator.userAgent (read by Google's auth JS) reflects the WebContents UA, - // not the request header, so the header-level Firefox switch in setupClientHintsOverride + // not the request header, so the header-level Firefox switch in setupGoogleAuthUserAgentOverride // must be matched here per navigation or the two layers disagree — itself a bot tell. // Restores the session's base identity off the auth hosts. Native-UA profiles opt out - // of the whole clean-UA path, so they keep their untouched identity everywhere. + // of the Firefox switch, so they keep their untouched identity everywhere. protected applyGoogleAuthUserAgent( guest: Electron.WebContents, url: string, @@ -24,8 +23,8 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility const browserPageId = this.tabIdByWebContentsId.get(guest.id) // Why: popup child windows get these policies but are never in tabIdByWebContentsId, so a direct // lookup misses the native-UA opt-out and would hand a native profile's popup the Firefox UA. - // That is worse than doing nothing: native sessions skip setupClientHintsOverride entirely, so - // the popup would send the raw Electron UA on the wire while navigator.userAgent claims Firefox. + // That is worse than doing nothing: native sessions never install the header-level Firefox + // switch, so the popup would send the Electron UA on the wire while navigator.userAgent claims Firefox. const ownerTabId = this.resolveBrowserTabIdForGuestWebContentsId(guest.id) // Session state is authoritative before renderer registration and after a native profile imports a source UA. const mode = @@ -56,15 +55,14 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // navigation (ERR_ABORTED) and replay the original request, which a POST-started OAuth chain // cannot survive — the sign-in lands on a blank tab. CDP retargets navigator.userAgent without // touching the navigation, and it outranks the WebContents UA from then on, so a guest that - // switches to it stays on it. The wire UA never depended on this write: setupClientHintsOverride - // rewrites User-Agent per request for auth-host URLs on its own. + // switches to it stays on it. The wire UA never depended on this write: + // setupGoogleAuthUserAgentOverride rewrites User-Agent per request for auth-host URLs on its own. if (options.duringRedirect === true || overrideState !== undefined) { if (this.canOverrideUserAgentOverCdp(guest)) { authOverrideIssuedOverCdp = true // Why: go through the viewport builder rather than writing nextUa raw, so both CDP writers - // resolve one identity for this URL — Firefox on auth hosts, the profile's clean base off - // them, any mobile preset preserved. Writing the session UA directly would put the - // unlaundered Electron token back on the wire. + // resolve one identity for this URL — Firefox on auth hosts, the session's base identity + // off them, any mobile preset preserved. void this.applyAuthUserAgentOverrideOverCdp( guest, (browserPageId ? this.viewportUaOverrideMobileByTabId.get(browserPageId) : undefined) ?? @@ -188,7 +186,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: Emulation.setUserAgentOverride is set once and stands across every later navigation, // outranking setUserAgent for navigator.userAgent. A viewport preset applied before reaching an - // auth host would otherwise pin navigator.userAgent to the Chrome-shaped preset UA while the + // auth host would otherwise pin navigator.userAgent to the session's preset UA while the // request header says Firefox — the two-layer disagreement this scope exists to remove. protected reapplyViewportUserAgentOverride( guest: Electron.WebContents, @@ -222,7 +220,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: the session UA is the profile's stable base identity. guest.getUserAgent() is not: // applyGoogleAuthUserAgent leaves it pinned to the Firefox auth UA once a guest switches to // the CDP override, so reading it back here would republish that identity on ordinary hosts. - baseUserAgent: cleanElectronUserAgent(baseUserAgent ?? guest.session.getUserAgent()) + baseUserAgent: baseUserAgent ?? guest.session.getUserAgent() }) ) } diff --git a/src/main/browser/browser-manager-state.ts b/src/main/browser/browser-manager-state.ts index bc65cc3d2dc..54b0f99b3c1 100644 --- a/src/main/browser/browser-manager-state.ts +++ b/src/main/browser/browser-manager-state.ts @@ -1,4 +1,3 @@ -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { BrowserGrabSessionController } from './browser-grab-session-controller' import type { BrowserCertificateTrustController } from './browser-certificate-trust-controller' import { @@ -190,56 +189,19 @@ export abstract class BrowserManagerState extends BrowserManagerViewportScrollSt this.settingsResolver = resolver } - // Why: addScriptToEvaluateOnNewDocument (CDP) is the only reliable pre-page-script hook per nav; executeJavaScript ran on the old page context. - protected injectAntiDetection(guest: Electron.WebContents): () => void { - let disposed = false - let reattachTimer: ReturnType | null = null - - const attach = (): void => { - if (disposed || guest.isDestroyed()) { - return - } - try { - if (!guest.debugger.isAttached()) { - guest.debugger.attach('1.3') - } - void guest.debugger - .sendCommand('Page.enable', {}) - .then(() => - guest.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) - ) - .catch(() => {}) - } catch { - /* best-effort — debugger may be unavailable */ - } - } - - // Why: proxy/bridge stop detaches the debugger and drops injections; re-attach (500ms delay to avoid racing a mid-restart) to keep overrides. + // Why: a debugger detach clears every CDP override Chromium holds, including the Google auth-host + // UA override, so the confirmed-override record must be dropped or the next auth navigation + // believes the identity is still installed and skips the write. + protected trackDebuggerDetachForAuthUserAgent(guest: Electron.WebContents): () => void { const onDetach = (): void => { this.authUserAgentOverrideStateByGuestId.delete(guest.id) - if (!disposed && !guest.isDestroyed() && reattachTimer === null) { - reattachTimer = setTimeout(() => { - reattachTimer = null - attach() - }, 500) - } } - try { - attach() guest.debugger.on('detach', onDetach) } catch { - /* best-effort */ + /* debugger may be unavailable */ } - return () => { - disposed = true - if (reattachTimer !== null) { - clearTimeout(reattachTimer) - reattachTimer = null - } try { guest.debugger.off('detach', onDetach) } catch { diff --git a/src/main/browser/browser-manager-types.ts b/src/main/browser/browser-manager-types.ts index a1b832a65bc..b91e2b741fe 100644 --- a/src/main/browser/browser-manager-types.ts +++ b/src/main/browser/browser-manager-types.ts @@ -117,7 +117,7 @@ export type PopupOwnerContext = { /** * What a guest is allowed to be. A browsing guest is the web — popups, clicked-link routing and - * anti-detection all apply. A workspace-document guest renders one granted document and gets none + * auth-identity tracking all apply. A workspace-document guest renders one granted document and gets none * of that; `host` is the renderer that minted its grant, and the only sink for what it reports. */ export type BrowserGuestPolicy = diff --git a/src/main/browser/browser-manager-viewport-override.test.ts b/src/main/browser/browser-manager-viewport-override.test.ts index b7d3bbabe0a..0ffc3c2a6e1 100644 --- a/src/main/browser/browser-manager-viewport-override.test.ts +++ b/src/main/browser/browser-manager-viewport-override.test.ts @@ -49,7 +49,6 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_CLEAN_UA, GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' @@ -207,7 +206,7 @@ describe('browserManager', () => { mobile: false }) expect(debuggerSendCommand).toHaveBeenLastCalledWith('Emulation.setUserAgentOverride', { - userAgent: GUEST_CLEAN_UA + userAgent: GUEST_ELECTRON_UA }) // Navigating to the auth host must move the standing override to the Firefox identity. @@ -218,11 +217,11 @@ describe('browserManager', () => { userAgent: googleAuthUserAgent() }) - // Leaving the auth host restores the clean Chrome-shaped preset UA. + // Leaving the auth host restores the session's own preset UA. debuggerSendCommand.mockClear() willRedirect({ preventDefault: vi.fn() }, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) // Why: not an ordering race — debugger.sendCommand dispatches in call order over one channel, so @@ -241,9 +240,9 @@ describe('browserManager', () => { } // Why mobile: on the desktop branch the break is masked by coincidence — applyGoogleAuthUserAgent - // has already switched the WebContents UA to Firefox, and cleanElectronUserAgent passes a Firefox - // UA through untouched, so the stale-URL desktop path happens to emit Firefox anyway. The mobile - // branch derives a Chrome-shaped iPhone UA from that same base and exposes the real defect. + // has already switched the WebContents UA to Firefox, so the stale-URL desktop path happens to + // emit Firefox anyway. The mobile branch derives a Chrome-shaped iPhone UA from the session base + // and exposes the real defect. it('does not leave the Chrome preset UA standing when a mobile preset lands mid-navigation onto an auth host', async () => { const { guest, debuggerSendCommand } = makeGuest(4251, 'https://example.com/') // Hold the preset's first CDP command open so the navigation lands inside its await window. @@ -332,7 +331,7 @@ describe('browserManager', () => { // Without the fix the resuming preset re-reads getURL() as the auth host and clobbers the // navigation's correct write, stranding the Firefox UA on a non-auth page. - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('falls back to the committed URL once a navigation commits or fails', async () => { @@ -378,7 +377,7 @@ describe('browserManager', () => { await flushViewportOps() expect(guest.setUserAgent).toHaveBeenLastCalledWith(GUEST_ELECTRON_UA) - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) // A later preset must also resolve the committed, non-auth URL. debuggerSendCommand.mockClear() @@ -457,7 +456,7 @@ describe('browserManager', () => { expect(guest.setUserAgent).not.toHaveBeenCalled() expect(debuggerSendCommand).not.toHaveBeenCalledWith( 'Emulation.setUserAgentOverride', - expect.objectContaining({ userAgent: GUEST_CLEAN_UA }) + expect.objectContaining({ userAgent: GUEST_ELECTRON_UA }) ) }) @@ -517,7 +516,7 @@ describe('browserManager', () => { didFailLoad(null, -3, 'Aborted', 'https://accounts.google.com/redirected', true) await flushViewportOps() expect(guest.setUserAgent).not.toHaveBeenCalled() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('preserves the auth identity when a viewport preset is cleared after a redirect', async () => { @@ -592,7 +591,7 @@ describe('browserManager', () => { didStartNavigation(null, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('reapplies a preset when navigation starts during its final UA write', async () => { @@ -849,8 +848,7 @@ describe('browserManager', () => { expect(debuggerAttach).toHaveBeenCalledWith('1.3') expect(debuggerSendCommand).toHaveBeenCalled() - // Why: detaching would clear Page.addScriptToEvaluateOnNewDocument - // (anti-detection). Guard regression. + // Why: detaching would clear every standing CDP override (viewport, auth UA). Guard regression. expect((guest.debugger as { detach?: unknown }).detach ?? undefined).toBeUndefined() }) diff --git a/src/main/browser/browser-manager-viewport-test-fixtures.ts b/src/main/browser/browser-manager-viewport-test-fixtures.ts index 8d68977f62e..076524ce5d0 100644 --- a/src/main/browser/browser-manager-viewport-test-fixtures.ts +++ b/src/main/browser/browser-manager-viewport-test-fixtures.ts @@ -3,8 +3,6 @@ import type { BrowserManagerMocks } from './browser-manager-test-harness' export const GUEST_ELECTRON_UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/134.0.0.0 Electron/30.0.0 Safari/537.36' -export const GUEST_CLEAN_UA = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/134.0.0.0 Safari/537.36' // Why: viewport UA writes are queued on the per-tab chain, so draining it takes more than one // microtask hop; loop until the chain is empty rather than guessing a tick count. @@ -53,7 +51,9 @@ export function createViewportGuestFactory( debugger: { isAttached: debuggerIsAttached, attach: debuggerAttach, - sendCommand: debuggerSendCommand + sendCommand: debuggerSendCommand, + on: vi.fn(), + off: vi.fn() } } return { diff --git a/src/main/browser/browser-manager-viewport.ts b/src/main/browser/browser-manager-viewport.ts index 1e79e760942..ce31dbe37e1 100644 --- a/src/main/browser/browser-manager-viewport.ts +++ b/src/main/browser/browser-manager-viewport.ts @@ -30,7 +30,7 @@ export abstract class BrowserManagerViewport extends BrowserManagerDownloadLifec return true } - // Why: emulate viewport via CDP; never detach the debugger here or per-guest overrides (addScriptToEvaluateOnNewDocument) are cleared. + // Why: emulate viewport via CDP; never detach the debugger here or the agent bridge's per-guest state is cleared. async setViewportOverride( browserTabId: string, override: BrowserViewportOverride | null diff --git a/src/main/browser/browser-session-partition-policies.test.ts b/src/main/browser/browser-session-partition-policies.test.ts index 78ce34d95fd..952c199536c 100644 --- a/src/main/browser/browser-session-partition-policies.test.ts +++ b/src/main/browser/browser-session-partition-policies.test.ts @@ -82,8 +82,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: async () => false })) vi.mock('./browser-session-ua', () => ({ - cleanElectronUserAgent: (userAgent: string) => userAgent, - setupClientHintsOverride: vi.fn() + setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ setBrowserSessionUserAgentMode: vi.fn() diff --git a/src/main/browser/browser-session-partition-policies.ts b/src/main/browser/browser-session-partition-policies.ts index 9f25d8840a2..ec3c68fb45e 100644 --- a/src/main/browser/browser-session-partition-policies.ts +++ b/src/main/browser/browser-session-partition-policies.ts @@ -9,7 +9,7 @@ import { } from './browser-session-proxy' import { hasSystemMediaAccess, requestSystemMediaAccess } from './browser-media-access' import { isAutoGrantedBrowserSessionPermission } from './browser-session-permission-policy' -import { cleanElectronUserAgent, setupClientHintsOverride } from './browser-session-ua' +import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { allowsBrowserWebAuthnPermission, @@ -92,10 +92,8 @@ export function installBrowserSessionPartitionPolicies( } browserManager.installCertificateRequestGuard(sess) - if (profile.userAgentMode !== 'native' && typeof sess.getUserAgent === 'function') { - const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) - sess.setUserAgent(cleanUA) - setupClientHintsOverride(sess, cleanUA) + if (profile.userAgentMode !== 'native') { + setupGoogleAuthUserAgentOverride(sess) } if (options?.permissions === 'deny') { sess.setPermissionRequestHandler((_webContents, _permission, callback) => callback(false)) @@ -191,11 +189,7 @@ export function applyBrowserSessionUserAgentModes(profiles: BrowserSessionProfil if (profile.userAgentMode === 'native') { continue } - - // Why: the default Electron UA leaks "Electron/X.X.X" + app name, which trips Cloudflare Turnstile. - const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) - sess.setUserAgent(cleanUA) - setupClientHintsOverride(sess, cleanUA) + setupGoogleAuthUserAgentOverride(sess) } catch { /* session not available yet (e.g. unit tests or pre-ready) */ } diff --git a/src/main/browser/browser-session-partition-proxy-install.test.ts b/src/main/browser/browser-session-partition-proxy-install.test.ts index 0841ed94785..4beeaab04c9 100644 --- a/src/main/browser/browser-session-partition-proxy-install.test.ts +++ b/src/main/browser/browser-session-partition-proxy-install.test.ts @@ -44,8 +44,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: vi.fn(async () => false) })) vi.mock('./browser-session-ua', () => ({ - cleanElectronUserAgent: vi.fn((ua: string) => ua), - setupClientHintsOverride: vi.fn() + setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ setBrowserSessionUserAgentMode: vi.fn(), diff --git a/src/main/browser/browser-session-registry.persistence.test.ts b/src/main/browser/browser-session-registry.persistence.test.ts index fdba71e16d6..67653c0c76c 100644 --- a/src/main/browser/browser-session-registry.persistence.test.ts +++ b/src/main/browser/browser-session-registry.persistence.test.ts @@ -27,7 +27,7 @@ function installModuleMocks( copyFailures = new Set() ): { sessionFromPartitionMock: ReturnType - setupClientHintsOverrideMock: ReturnType + setupGoogleAuthUserAgentOverrideMock: ReturnType browserManagerHandleGuestWillDownloadMock: ReturnType browserManagerNotifyPermissionDeniedMock: ReturnType requestSystemMediaAccessMock: ReturnType @@ -36,6 +36,7 @@ function installModuleMocks( partition, setUserAgent: vi.fn(), getUserAgent: vi.fn(() => 'Mozilla/5.0 Electron/31 Orca'), + webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -45,7 +46,7 @@ function installModuleMocks( clearStorageData: vi.fn().mockResolvedValue(undefined), clearCache: vi.fn().mockResolvedValue(undefined) })) - const setupClientHintsOverrideMock = vi.fn() + const setupGoogleAuthUserAgentOverrideMock = vi.fn() const browserManagerHandleGuestWillDownloadMock = vi.fn() const browserManagerNotifyPermissionDeniedMock = vi.fn() const requestSystemMediaAccessMock = vi.fn().mockResolvedValue(true) @@ -119,8 +120,7 @@ function installModuleMocks( requestSystemMediaAccess: requestSystemMediaAccessMock })) vi.doMock('./browser-session-ua', () => ({ - cleanElectronUserAgent: vi.fn((ua: string) => ua.replace(/\s*Electron\/\S+/, '')), - setupClientHintsOverride: setupClientHintsOverrideMock + setupGoogleAuthUserAgentOverride: setupGoogleAuthUserAgentOverrideMock })) // This suite models replay with an in-memory filesystem. The real file-backed SQLite merge has // dedicated coverage; these fixtures are legacy unmarked images and keep the copy path. @@ -149,7 +149,7 @@ function installModuleMocks( return { sessionFromPartitionMock, - setupClientHintsOverrideMock, + setupGoogleAuthUserAgentOverrideMock, browserManagerHandleGuestWillDownloadMock, browserManagerNotifyPermissionDeniedMock, requestSystemMediaAccessMock @@ -234,21 +234,24 @@ describe('BrowserSessionRegistry persistence', () => { }) }) - it('keeps UA cleaning as the fallback for profiles without an override', async () => { + // Why: the stock Electron UA is what clears Cloudflare; only the Google auth switch installs. + it('keeps the stock UA and installs the Google auth switch for profiles without an override', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Default identity') const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value - expect(profileSession.setUserAgent).toHaveBeenCalledWith('Mozilla/5.0 Orca') - expect(setupClientHintsOverrideMock).toHaveBeenCalledWith(profileSession, 'Mozilla/5.0 Orca') + expect(profileSession.setUserAgent).not.toHaveBeenCalled() + expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalledWith(profileSession) }) it('leaves UA and client hints untouched for native-mode profiles', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Google', { userAgentMode: 'native' }) @@ -256,7 +259,7 @@ describe('BrowserSessionRegistry persistence', () => { const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value const { getBrowserSessionUserAgentMode } = await import('./browser-session-user-agent-mode') expect(profileSession.setUserAgent).not.toHaveBeenCalled() - expect(setupClientHintsOverrideMock).not.toHaveBeenCalled() + expect(setupGoogleAuthUserAgentOverrideMock).not.toHaveBeenCalled() expect(getBrowserSessionUserAgentMode(profileSession as never)).toBe('native') }) @@ -379,7 +382,7 @@ describe('BrowserSessionRegistry persistence', () => { // Why: imports before Aug 2026 persisted a synthesized source-browser UA // (fork imports as a broken Chrome/1.x, Chrome imports as a valid version). // Neither may ever be applied again — the engine-derived UA is the only one. - it('ignores legacy persisted UAs, valid or broken, and applies the engine UA', async () => { + it('ignores legacy persisted UAs, valid or broken, and keeps the engine UA', async () => { const importedPartition = 'persist:orca-browser-session-11111111-1111-4111-8111-111111111111' const brokenUa = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/1.158.1 Safari/537.36' @@ -405,7 +408,8 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -413,16 +417,9 @@ describe('BrowserSessionRegistry persistence', () => { const appliedUas = sessionFromPartitionMock.mock.results.flatMap((r) => r.value.setUserAgent.mock.calls.map((c: unknown[]) => c[0]) ) - expect(appliedUas).not.toContain(brokenUa) - expect(appliedUas).not.toContain(validUa) - // Why: every non-native profile falls to Orca's own cleaned engine UA. - expect(appliedUas.length).toBeGreaterThan(0) - expect(appliedUas.every((ua) => ua === 'Mozilla/5.0 Orca')).toBe(true) - expect( - setupClientHintsOverrideMock.mock.calls.every( - (c: unknown[]) => c[1] !== brokenUa && c[1] !== validUa - ) - ).toBe(true) + // Why: no persisted UA is ever written back; every profile keeps the engine's stock UA. + expect(appliedUas).toEqual([]) + expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalled() }) it('never applies a legacy persisted UA to a native-mode profile', async () => { @@ -487,7 +484,8 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -498,7 +496,7 @@ describe('BrowserSessionRegistry persistence', () => { expect(importedSessions.length).toBeGreaterThan(0) expect(importedSessions.every((sess) => sess.setUserAgent.mock.calls.length === 0)).toBe(true) expect( - setupClientHintsOverrideMock.mock.calls.some( + setupGoogleAuthUserAgentOverrideMock.mock.calls.some( ([sess]) => (sess as { partition?: string }).partition === importedPartition ) ).toBe(false) diff --git a/src/main/browser/browser-session-registry.test.ts b/src/main/browser/browser-session-registry.test.ts index d5111483fcf..ae81ffda4e2 100644 --- a/src/main/browser/browser-session-registry.test.ts +++ b/src/main/browser/browser-session-registry.test.ts @@ -33,7 +33,7 @@ vi.mock('./browser-manager', () => ({ import { browserSessionRegistry } from './browser-session-registry' import { googleAuthUserAgent } from './browser-google-auth-ua' -import { setupClientHintsOverride } from './browser-session-ua' +import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserNetworkProxySettingsResolver } from './browser-session-proxy' import { handleElectronProxyLogin } from '../network/electron-proxy-credentials' import { applyProxySettingsToSession } from '../network/proxy-settings' @@ -54,6 +54,7 @@ describe('BrowserSessionRegistry', () => { askForMediaAccessMock.mockResolvedValue(true) getMediaAccessStatusMock.mockReturnValue('granted') sessionFromPartitionMock.mockReturnValue({ + webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -528,80 +529,46 @@ describe('BrowserSessionRegistry', () => { }) }) - describe('setupClientHintsOverride', () => { - it('overrides sec-ch-ua headers for Edge UA', () => { + describe('setupGoogleAuthUserAgentOverride', () => { + const STOCK_UA = + 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/147.0.6890.3 Electron/43.0.0 Safari/537.36' + + function install(): (details: unknown, callback: ReturnType) => void { const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const edgeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36 Edg/147.0.3210.5' - - setupClientHintsOverride(mockSess, edgeUa) - + setupGoogleAuthUserAgentOverride({ webRequest: { onBeforeSendHeaders } } as never) expect(onBeforeSendHeaders).toHaveBeenCalledWith( { urls: ['https://*/*'] }, expect.any(Function) ) + return onBeforeSendHeaders.mock.calls[0][1] + } + // Why: the Electron token is what clears Cloudflare Turnstile; a Chrome-shaped UA with no + // client hints is what it rejects, so ordinary hosts must see the session's UA untouched. + it('leaves the stock Electron UA and its client hints alone off the auth hosts', () => { + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] listener( - { requestHeaders: { 'sec-ch-ua': 'old', 'sec-ch-ua-full-version-list': 'old' } }, + { + url: 'https://example.com/api', + requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', Cookie: 'abc=123' } + }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['sec-ch-ua']).toContain('Microsoft Edge') - expect(modified['sec-ch-ua']).toContain('"147"') - expect(modified['sec-ch-ua-full-version-list']).toContain('147.0.3210.5') - }) - - it('overrides sec-ch-ua headers for Chrome UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - - setupClientHintsOverride(mockSess, chromeUa) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener({ requestHeaders: { 'sec-ch-ua': 'old' } }, callback) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['sec-ch-ua']).toContain('Google Chrome') - expect(modified['sec-ch-ua']).not.toContain('Microsoft Edge') - }) - - it('registers handler even for non-Chrome UA but leaves sec-ch-ua untouched off auth hosts', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - - // Why: the Google-auth Firefox switch must install regardless of the base UA. - setupClientHintsOverride(mockSess, 'Mozilla/5.0 (compatible; MSIE 10.0)') - - expect(onBeforeSendHeaders).toHaveBeenCalledWith( - { urls: ['https://*/*'] }, - expect.any(Function) - ) - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener({ url: 'https://example.com/', requestHeaders: { 'sec-ch-ua': 'old' } }, callback) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toBe('old') + expect(modified['User-Agent']).toBe(STOCK_UA) + expect(modified['sec-ch-ua']).toBe('old') + expect(modified.Cookie).toBe('abc=123') }) it('presents a Firefox UA and strips client hints on Google auth hosts', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] listener( { url: 'https://accounts.google.com/v3/signin/identifier', requestHeaders: { - 'User-Agent': 'Chrome/147', + 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', 'sec-ch-ua-full-version-list': 'old', 'sec-ch-ua-platform': '"macOS"' @@ -610,7 +577,7 @@ describe('BrowserSessionRegistry', () => { callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toMatch(/Firefox\/\d/) + expect(modified['User-Agent']).toBe(googleAuthUserAgent()) expect(modified['User-Agent']).not.toContain('Chrome') expect(modified['sec-ch-ua']).toBeUndefined() expect(modified['sec-ch-ua-full-version-list']).toBeUndefined() @@ -618,15 +585,8 @@ describe('BrowserSessionRegistry', () => { }) it('strips client hints on a cross-host request that carries the Firefox auth UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] // Subresource/XHR to a non-auth Google host while the auth document is on // screen: the WebContents Firefox UA leaks onto the request header. listener( @@ -651,99 +611,19 @@ describe('BrowserSessionRegistry', () => { expect(modified['sec-ch-ua-mobile']).toBeUndefined() }) - it('keeps the clean Chrome identity on cross-host requests that carry the Chrome UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, chromeUa) - + it('keeps the session identity on Google app subdomains (not auth hosts)', () => { + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - // Regression guard: non-Google sites (Cloudflare) must keep Chrome hints. listener( { - url: 'https://example.com/api', - requestHeaders: { 'User-Agent': chromeUa, 'sec-ch-ua': 'old' } - }, - callback - ) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('does not strip hints for the Firefox UA when googleAuthOverride is disabled', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, chromeUa, { googleAuthOverride: false }) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { - url: 'https://play.google.com/log', - requestHeaders: { 'User-Agent': googleAuthUserAgent(), 'sec-ch-ua': 'old' } - }, - callback - ) - // Imported-native profiles never install the Firefox switch, so the strip - // branch stays inert and hints are aligned to Chrome instead. - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('keeps Chrome client hints on Google app subdomains (not auth hosts)', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { url: 'https://myaccount.google.com/', requestHeaders: { 'sec-ch-ua': 'old' } }, - callback - ) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('keeps an imported native UA on auth hosts while aligning its Chrome hints', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const importedUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, importedUa, { googleAuthOverride: false }) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { - url: 'https://accounts.google.com/v3/signin/identifier', - requestHeaders: { 'User-Agent': importedUa, 'sec-ch-ua': 'old' } + url: 'https://myaccount.google.com/', + requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old' } }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toBe(importedUa) - expect(modified['sec-ch-ua']).toContain('Google Chrome') - }) - - it('leaves non-Client-Hints headers unchanged', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride(mockSess, 'Mozilla/5.0 Chrome/147.0.0.0 Safari/537.36') - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { requestHeaders: { Cookie: 'abc=123', 'sec-ch-ua': 'old', Accept: 'text/html' } }, - callback - ) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified.Cookie).toBe('abc=123') - expect(modified.Accept).toBe('text/html') + expect(modified['User-Agent']).toBe(STOCK_UA) + expect(modified['sec-ch-ua']).toBe('old') }) }) }) diff --git a/src/main/browser/browser-session-ua-wire-identity.electron.test.ts b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts new file mode 100644 index 00000000000..4e0719b2745 --- /dev/null +++ b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts @@ -0,0 +1,180 @@ +import { spawnSync } from 'node:child_process' +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { build as buildVite } from 'vite' + +// Why this runs a real Electron: Cloudflare Turnstile rejects a Chrome-shaped UA that ships no +// client hints (error 600010) and clears a declared Electron client. The header layer is the +// only place that identity can be proven, and the vm-based unit tests cannot see Chromium's +// header emission at all. Every partition must therefore keep the stock Electron UA on the wire +// for ordinary hosts and present the Firefox identity on Google's sign-in hosts only. + +const electronBinary = createRequire(import.meta.url)('electron') as string +const fixtureRoots: string[] = [] + +afterAll(() => { + for (const root of fixtureRoots) { + rmSync(root, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }) + } +}) + +// Retry once when Electron startup times out before `ready`; keep later failures fatal. +const FIXTURE_LAUNCH_ATTEMPTS = 2 + +type CapturedRequest = { + url: string + userAgent: string | null + clientHints: string[] +} + +type FixtureResult = { + sessionUserAgent: string + navigatorUserAgent: string + requests: CapturedRequest[] +} + +function neverReachedElectronReady(fixtureResult: string): boolean { + try { + return (JSON.parse(fixtureResult) as { step?: string }).step === 'timed out after starting' + } catch { + return false + } +} + +function buildFixtureMain(modulePath: string, resultPath: string): string { + return ` +const { app, BrowserWindow, session } = require('electron') +const { writeFileSync } = require('node:fs') +const { setupGoogleAuthUserAgentOverride } = require(${JSON.stringify(modulePath)}) +const resultPath = ${JSON.stringify(resultPath)} +let currentStep = 'starting' +const mark = (step) => { + currentStep = step + writeFileSync(resultPath, JSON.stringify({ step })) +} + +async function run() { + const timeout = setTimeout(() => { + writeFileSync(resultPath, JSON.stringify({ step: 'timed out after ' + currentStep })) + app.exit(1) + }, 15000) + await app.whenReady() + mark('ready') + const partition = 'persist:wire-identity-test' + const sess = session.fromPartition(partition) + setupGoogleAuthUserAgentOverride(sess) + mark('auth switch installed') + + // Why: onSendHeaders reports the headers exactly as they leave the network stack, after the + // product's onBeforeSendHeaders listener has rewritten them. The requests must actually be + // dispatched for it to fire, so the session is pointed at a proxy that refuses every + // connection: nothing reaches the real hosts and every load fails fast. + await sess.setProxy({ proxyRules: 'http://127.0.0.1:9', proxyBypassRules: '<-loopback>' }) + const requests = [] + sess.webRequest.onSendHeaders({ urls: ['https://*/*'] }, (details) => { + const headers = details.requestHeaders || {} + const uaKey = Object.keys(headers).find((key) => key.toLowerCase() === 'user-agent') + requests.push({ + url: details.url, + userAgent: uaKey ? headers[uaKey] : null, + clientHints: Object.keys(headers) + .filter((key) => key.toLowerCase().startsWith('sec-ch-ua')) + .sort() + }) + }) + + const window = new BrowserWindow({ show: false, webPreferences: { partition } }) + mark('window created') + for (const url of ['https://example.com/', 'https://accounts.google.com/v3/signin/identifier']) { + await window.loadURL(url).catch(() => {}) + } + mark('navigations attempted') + const navigatorUserAgent = await window.webContents.executeJavaScript('navigator.userAgent') + clearTimeout(timeout) + writeFileSync(resultPath, JSON.stringify({ + sessionUserAgent: sess.getUserAgent(), + navigatorUserAgent, + requests + })) + window.destroy() + app.exit(0) +} + +run().catch((error) => { + writeFileSync(resultPath, JSON.stringify({ step: currentStep, error: String(error?.stack || error) })) + app.exit(1) +}) +` +} + +async function runFixture(): Promise { + const root = mkdtempSync(join(tmpdir(), 'orca-wire-identity-')) + fixtureRoots.push(root) + const modulePath = join(root, 'browser-session-ua.cjs') + const resultPath = join(root, 'result.json') + const fixturePath = join(root, 'main.cjs') + await buildVite({ + configFile: false, + logLevel: 'silent', + build: { + emptyOutDir: false, + lib: { + entry: join(process.cwd(), 'src/main/browser/browser-session-ua.ts'), + formats: ['cjs'], + fileName: () => 'browser-session-ua.cjs' + }, + outDir: root, + target: 'node20', + rollupOptions: { external: ['electron', /^node:/] } + } + }) + writeFileSync(fixturePath, buildFixtureMain(modulePath, resultPath)) + const { ELECTRON_RUN_AS_NODE: _electronRunAsNode, ...env } = process.env + const executable = process.platform === 'linux' ? 'xvfb-run' : electronBinary + for (let attempt = 1; ; attempt += 1) { + rmSync(resultPath, { force: true }) + // Why a fresh profile per attempt: a launch that never reached `ready` may have left the + // Chromium profile mid-initialization, and reusing it would bias the retry. + const electronArgs = [fixturePath, `--user-data-dir=${join(root, `profile-${attempt}`)}`] + const run = spawnSync( + executable, + process.platform === 'linux' + ? ['--auto-servernum', electronBinary, ...electronArgs, '--no-sandbox'] + : electronArgs, + { encoding: 'utf8', env, timeout: 60_000 } + ) + const fixtureResult = existsSync(resultPath) ? readFileSync(resultPath, 'utf8') : 'no result' + if (attempt < FIXTURE_LAUNCH_ATTEMPTS && neverReachedElectronReady(fixtureResult)) { + continue + } + expect(run.error).toBeUndefined() + expect(run.status, `${fixtureResult}\n${run.stdout}\n${run.stderr}`).toBe(0) + return JSON.parse(fixtureResult) as FixtureResult + } +} + +describe('browser session wire identity under Electron', () => { + it('sends the stock Electron UA to ordinary hosts and Firefox to Google auth hosts', async () => { + const result = await runFixture() + + // Presence precondition: the stock identity still carries the Electron token that the old + // Chrome-shaped rewrite stripped, so an identity check below cannot pass on an empty UA. + expect(result.sessionUserAgent).toMatch(/ Electron\/\d/) + + const ordinary = result.requests.find((request) => request.url === 'https://example.com/') + expect(ordinary, JSON.stringify(result.requests)).toBeDefined() + expect(ordinary?.userAgent).toBe(result.sessionUserAgent) + expect(result.navigatorUserAgent).toBe(result.sessionUserAgent) + + const auth = result.requests.find((request) => + request.url.startsWith('https://accounts.google.com/') + ) + expect(auth, JSON.stringify(result.requests)).toBeDefined() + expect(auth?.userAgent).toMatch(/Firefox\/\d/) + expect(auth?.userAgent).not.toContain('Chrome') + expect(auth?.clientHints).toEqual([]) + }) +}) diff --git a/src/main/browser/browser-session-ua.ts b/src/main/browser/browser-session-ua.ts index 96c55cf5ac9..1375ebc66f5 100644 --- a/src/main/browser/browser-session-ua.ts +++ b/src/main/browser/browser-session-ua.ts @@ -8,93 +8,28 @@ import { stripClientHints } from './browser-google-auth-ua' -// Why: Electron's default UA includes "Electron/X.X.X" and the app name -// (e.g. "orca/1.2.3"), which Cloudflare Turnstile and other bot detectors -// flag as non-human traffic. Strip those tokens so the webview's UA and -// sec-ch-ua Client Hints look like standard Chrome. -export function cleanElectronUserAgent(ua: string): string { - return ( - ua - .replace(/\s+Electron\/\S+/, '') - // Why: \S+ matches any non-whitespace token (e.g. "orca/1.3.8-rc.0") - // including pre-release semver strings that [\d.]+ would miss. - .replace(/(\)\s+)\S+\s+(Chrome\/)/, '$1$2') - ) -} - -// Why: Electron emits sec-ch-ua brands like "Not A(Brand" without a -// "Google Chrome" entry, which disagrees with the Chrome-shaped UA the session -// presents. Rewrite the hint headers to the brand set Chrome ships for the same -// engine version so the two surfaces tell one story. Also owns the Google -// auth-host Firefox switch, which must install even for a non-Chrome-shaped UA. -export function setupClientHintsOverride( - sess: Session, - ua: string, - options: { googleAuthOverride?: boolean } = {} -): void { - // Why: only Chrome-shaped base UAs carry sec-ch-ua hints to rewrite, but the - // Google-auth Firefox switch below must install regardless, so keep the hints - // optional rather than bailing out of the whole handler. - const chromeHints = buildChromeClientHints(ua) +// Why: the session keeps Electron's stock UA. Stripping the Electron/app tokens to look like +// plain Chrome is what Cloudflare Turnstile rejects (error 600010): a Chrome UA that ships no +// client hints reads as a spoof, while a declared Electron client clears the same challenge. +// This handler only owns the Google auth-host Firefox switch, which is a proven, host-scoped +// exception that must stay consistent across the header and every cross-host subresource. +export function setupGoogleAuthUserAgentOverride(sess: Session): void { const firefoxUa = googleAuthUserAgent() sess.webRequest.onBeforeSendHeaders({ urls: ['https://*/*'] }, (details, callback) => { const headers = details.requestHeaders - if (options.googleAuthOverride !== false && isGoogleAuthUrl(details.url)) { + if (isGoogleAuthUrl(details.url)) { // Why: present a Firefox identity on Google's sign-in hosts so the user logs // in inside the app and Google issues self-refreshing bound cookies. Strip // sec-ch-ua* because real Firefox sends none. setUserAgentHeader(headers, firefoxUa) stripClientHints(headers) - callback({ requestHeaders: headers }) - return - } - if (options.googleAuthOverride !== false && currentUserAgent(headers) === firefoxUa) { - // Why: while the auth document is on screen the WebContents UA is Firefox, - // so its cross-host subresource/XHR requests (gstatic, play.google.com, the - // sign-in challenge endpoints) reach here carrying the Firefox UA yet still - // bearing Chromium client hints. Rewriting those to Chrome pairs a Firefox - // UA with Chrome hints — a sharper cross-host identity tell than either - // alone, which can stall Google's password-submit challenge. Real Firefox - // sends no client hints, so strip them to keep one identity for the flow. + } else if (currentUserAgent(headers) === firefoxUa) { + // Why: while the auth document is on screen the WebContents UA is Firefox, so its + // cross-host subresource/XHR requests carry the Firefox UA yet still bear Chromium + // client hints — a sharper cross-host identity tell than either alone. stripClientHints(headers) - callback({ requestHeaders: headers }) - return - } - if (chromeHints) { - for (const key of Object.keys(headers)) { - const lower = key.toLowerCase() - if (lower === 'sec-ch-ua') { - headers[key] = chromeHints.secChUa - } else if (lower === 'sec-ch-ua-full-version-list') { - headers[key] = chromeHints.secChUaFull - } - } } callback({ requestHeaders: headers }) }) } - -function buildChromeClientHints(ua: string): { secChUa: string; secChUaFull: string } | null { - const chromeMatch = ua.match(/Chrome\/([\d.]+)/) - if (!chromeMatch) { - return null - } - const fullChromeVersion = chromeMatch[1] - const majorVersion = fullChromeVersion.split('.')[0] - - let brand = 'Google Chrome' - let brandFullVersion = fullChromeVersion - - const edgeMatch = ua.match(/Edg\/([\d.]+)/) - if (edgeMatch) { - brand = 'Microsoft Edge' - brandFullVersion = edgeMatch[1] - } - const brandMajor = brandFullVersion.split('.')[0] - - return { - secChUa: `"${brand}";v="${brandMajor}", "Chromium";v="${majorVersion}", "Not/A)Brand";v="24"`, - secChUaFull: `"${brand}";v="${brandFullVersion}", "Chromium";v="${fullChromeVersion}", "Not/A)Brand";v="24.0.0.0"` - } -} diff --git a/src/main/browser/browser-viewport-user-agent.ts b/src/main/browser/browser-viewport-user-agent.ts index 7dedf8d8a6e..b8159a44c2a 100644 --- a/src/main/browser/browser-viewport-user-agent.ts +++ b/src/main/browser/browser-viewport-user-agent.ts @@ -23,7 +23,7 @@ export type ViewportUserAgentOverride = { } // Why: responsive sites UA-sniff; this is Chrome DevTools' default iPhone UA template with the real -// Chrome major spliced in to keep sec-ch-ua consistent (see setupClientHintsOverride). +// Chrome major spliced in so the userAgentMetadata brands below agree with it. function buildMobileUserAgent(chromeMajor: string): string { return `Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) CriOS/${chromeMajor}.0.0.0 Mobile/15E148 Safari/604.1` } @@ -44,7 +44,7 @@ export function buildViewportUserAgentOverride(args: { return { userAgent: googleAuthUserAgent() } } if (!args.mobile) { - // Why: desktop presets still need the clean (non-Electron) UA so Cloudflare/Turnstile don't flag the session. + // Why: desktop presets republish the session's own identity unchanged. return { userAgent: args.baseUserAgent } } const chromeMajor = extractChromeMajor(args.baseUserAgent) diff --git a/src/main/browser/browser-webauthn-profile-delete.test.ts b/src/main/browser/browser-webauthn-profile-delete.test.ts index 9a5e129885f..3c471fe2dc2 100644 --- a/src/main/browser/browser-webauthn-profile-delete.test.ts +++ b/src/main/browser/browser-webauthn-profile-delete.test.ts @@ -48,7 +48,8 @@ function mockSession(): MockSession { setDevicePermissionHandler: vi.fn(), setDisplayMediaRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), - setPermissionRequestHandler: vi.fn() + setPermissionRequestHandler: vi.fn(), + webRequest: { onBeforeSendHeaders: vi.fn() } }) as unknown as MockSession } diff --git a/src/main/browser/cdp-debugger-channel.ts b/src/main/browser/cdp-debugger-channel.ts index 352bc741ef0..18b819a1ce9 100644 --- a/src/main/browser/cdp-debugger-channel.ts +++ b/src/main/browser/cdp-debugger-channel.ts @@ -1,6 +1,5 @@ import { WebSocket } from 'ws' import type { WebContents } from 'electron' -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { acquireElectronDebugger, type ElectronDebuggerLease } from './electron-debugger-lease' import type { CdpClientResponseWriter } from './cdp-client-response-writer' import type { CdpSyntheticSessionRegistry } from './cdp-synthetic-session-registry' @@ -34,14 +33,8 @@ export class CdpDebuggerChannel { } this.attached = true - // Why: attaching the CDP debugger sets navigator.webdriver = true and - // exposes other automation signals that Cloudflare Turnstile checks. - // Inject before any page loads so challenges succeed. try { await this.webContents.debugger.sendCommand('Page.enable', {}) - await this.webContents.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) } catch { /* best-effort — page domain may not be ready yet */ } diff --git a/src/main/browser/cdp-debugger-events.ts b/src/main/browser/cdp-debugger-events.ts index 023d648a8ca..59d30105910 100644 --- a/src/main/browser/cdp-debugger-events.ts +++ b/src/main/browser/cdp-debugger-events.ts @@ -32,9 +32,11 @@ export function createCdpDebuggerMessageListener( | undefined if (p?.sessionId && p.targetInfo?.type === 'iframe' && p.targetInfo.targetId) { state.iframeSessions.set(p.targetInfo.targetId, p.sessionId) + // Why: no Runtime.enable here. Cross-origin iframes include challenge widgets + // (Cloudflare Turnstile), and the Runtime domain's console/Error.stack serialization + // is the CDP tell they detect; nothing reads iframe Runtime events anyway. guest.debugger.sendCommand('DOM.enable', {}, p.sessionId).catch(() => {}) guest.debugger.sendCommand('Accessibility.enable', {}, p.sessionId).catch(() => {}) - guest.debugger.sendCommand('Runtime.enable', {}, p.sessionId).catch(() => {}) } } if (method === 'Target.detachedFromTarget') { diff --git a/src/main/browser/cdp-debugger-lifecycle.ts b/src/main/browser/cdp-debugger-lifecycle.ts index f969f113d59..225eb689dba 100644 --- a/src/main/browser/cdp-debugger-lifecycle.ts +++ b/src/main/browser/cdp-debugger-lifecycle.ts @@ -1,5 +1,4 @@ import type { WebContents } from 'electron' -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { BrowserError } from './browser-error' import type { CdpTabState } from './cdp-auxiliary-commands' import type { CdpCommandSender } from './snapshot-engine' @@ -62,11 +61,6 @@ export class CdpDebuggerLifecycle { flatten: true }) - // Why: CDP attach exposes automation signals (navigator.webdriver) that Cloudflare checks; override per new document. - await sender('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) - // Why: only remove this bridge's listeners; screencast/proxy sessions share the debugger and own their teardown. this.removeDebuggerListeners(guest, state) diff --git a/src/main/browser/cdp-ws-proxy-focus-replay.test.ts b/src/main/browser/cdp-ws-proxy-focus-replay.test.ts index 71b00cd9ff6..8404614ef8d 100644 --- a/src/main/browser/cdp-ws-proxy-focus-replay.test.ts +++ b/src/main/browser/cdp-ws-proxy-focus-replay.test.ts @@ -52,7 +52,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 99 }], ['DOM.focus', { backendNodeId: 99 }], ['Input.insertText', { text: 'hello' }] @@ -80,7 +79,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.result).toEqual({}) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 123 }, 'oopif-session-123'], ['DOM.focus', { backendNodeId: 123 }, 'oopif-session-123'], ['Input.insertText', { text: 'frame text' }, 'oopif-session-123'] @@ -112,7 +110,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 44 }], ['Runtime.callFunctionOn', { functionDeclaration: '() => document.activeElement?.id' }], ['Input.insertText', { text: 'after eval' }] @@ -155,7 +152,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 55 }], ['Input.insertText', { text: 'fallback' }] ]) @@ -197,7 +193,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 77 }], ['DOM.focus', { backendNodeId: 77 }] ]) @@ -242,7 +237,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse?.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'DOM.focus', 'DOM.focus', 'Input.insertText' @@ -270,12 +264,7 @@ describe('CdpWsProxy DOM.focus replay', () => { // Why: both Page.bringToFront and Input.insertText natively call focus(), // independent of the (now-cleared) DOM.focus replay. expect(mock.webContents.focus).toHaveBeenCalledTimes(2) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'DOM.focus', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'DOM.focus', 'Input.insertText']) client.close() }) @@ -298,7 +287,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'DOM.focus', 'Page.captureScreenshot', 'Input.insertText' @@ -322,12 +310,7 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.id).toBe(34) expect(insertResponse.result).toEqual({}) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'DOM.focus', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'DOM.focus', 'Input.insertText']) second.close() }) diff --git a/src/main/browser/cdp-ws-proxy.test.ts b/src/main/browser/cdp-ws-proxy.test.ts index c5f098ffa91..d5a2d14a05b 100644 --- a/src/main/browser/cdp-ws-proxy.test.ts +++ b/src/main/browser/cdp-ws-proxy.test.ts @@ -400,11 +400,7 @@ describe('CdpWsProxy', () => { }) expect(mock.webContents.focus).toHaveBeenCalledTimes(1) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'Input.insertText']) client.close() }) @@ -421,7 +417,6 @@ describe('CdpWsProxy', () => { expect(response.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'Network.enable', 'Page.enable', 'Page.setLifecycleEventsEnabled', @@ -442,7 +437,6 @@ describe('CdpWsProxy', () => { expect(response.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'Network.enable', 'Page.enable', 'Page.setLifecycleEventsEnabled' @@ -462,7 +456,7 @@ describe('CdpWsProxy', () => { sessionId: 'iframe-session-123' }) - expect(getSendCommandCalls(mock).slice(2)).toEqual([ + expect(getSendCommandCalls(mock).slice(1)).toEqual([ ['Network.enable', {}, 'iframe-session-123'], ['Page.enable', {}, 'iframe-session-123'], ['Page.setLifecycleEventsEnabled', { enabled: true }, 'iframe-session-123'], @@ -481,7 +475,7 @@ describe('CdpWsProxy', () => { sessionId: 'iframe-session-123' }) - expect(getSendCommandCalls(mock).slice(2)).toEqual([ + expect(getSendCommandCalls(mock).slice(1)).toEqual([ ['Network.enable', {}, 'iframe-session-123'], ['Page.enable', {}, 'iframe-session-123'], ['Page.setLifecycleEventsEnabled', { enabled: true }, 'iframe-session-123'], @@ -562,11 +556,7 @@ describe('CdpWsProxy', () => { expect(response.id).toBe(13) expect(response.result).toEqual({}) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'Runtime.evaluate' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'Runtime.evaluate']) client.close() }) diff --git a/src/main/window/main-window-webview-security.ts b/src/main/window/main-window-webview-security.ts index da6e4159a58..a662449eb58 100644 --- a/src/main/window/main-window-webview-security.ts +++ b/src/main/window/main-window-webview-security.ts @@ -111,7 +111,7 @@ export function installMainWindowWebviewSecurity(mainWindow: BrowserWindow): voi mainWindow.webContents.on('did-attach-webview', (_event, guest) => { if (isDocPreviewSession(guest.session)) { - // Why: preview guests never join browser-tab routing, popups or anti-detection; the + // Why: preview guests never join browser-tab routing, popups or auth-identity tracking; the // workspace-doc profile is what refuses all three. The attach is also the point a live window // exists to receive read failures for that guest. setDocPreviewFailureSink(mainWindow.webContents) diff --git a/tests/tools/google-signin-ua-probe.cjs b/tests/tools/google-signin-ua-probe.cjs index 7adf06bb7a0..70c834b3d9d 100644 --- a/tests/tools/google-signin-ua-probe.cjs +++ b/tests/tools/google-signin-ua-probe.cjs @@ -10,8 +10,8 @@ const MODES = new Set([ 'electron-fixed', 'firefox-auth', 'firefox-fixed', - // Replicates the SHIPPED app exactly (setupClientHintsOverride + - // applyGoogleAuthUserAgent): Firefox UA is written to the WebContents on auth + // Replicates the app as it shipped before the UA rewrite was removed + // (cleaned Chrome-shaped session UA + the Google auth Firefox switch): Firefox UA is written to the WebContents on auth // navs and to the request header only for auth-host URLs; every other request // keeps whatever UA the WebContents carries. Logs incoming vs outgoing // identity for ALL requests to expose cross-host mismatches during the flow. @@ -185,7 +185,7 @@ app.whenReady().then(async () => { if (mode === 'app-fixed' && currentUa === identities.firefox) { removeClientHints(headers) } else { - // Real setupClientHintsOverride builds Chrome hints once from the + // The retired client-hints rewrite built Chrome hints once from the // session's cleaned UA (a closure), never from the per-request UA. applyChromeClientHints(headers, identities.cleaned) } From ba4bbacd6b209e0d67730ac1497bbd504457397f Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:18:21 -0400 Subject: [PATCH 06/26] fix(relay-ops): align the cloud-data freshness bar with Cloud Monitoring publish lag (#18798) --- .../src/incident-live-preflight-cli.test.ts | 15 +- .../src/incident-live-preflight-cli.ts | 7 +- .../relay-ops/src/incident-monitor-cli.ts | 2 + .../src/incident-monitor-sources.test.ts | 4 +- .../relay-ops/src/incident-monitor-sources.ts | 2 +- .../relay-ops/src/incident-monitor.test.ts | 258 +++++++++++++++++- cloud/apps/relay-ops/src/incident-monitor.ts | 105 ++++++- cloud/docs/relay-incident-monitor.md | 29 +- 8 files changed, 388 insertions(+), 34 deletions(-) diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts index 3252ec7645e..412c2905408 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts @@ -6,7 +6,10 @@ import { livePreflightGcloud, runIncidentLivePreflight } from './incident-live-preflight-cli.js' -import type { IncidentSample } from './incident-monitor.js' +import { + INCIDENT_MONITOR_THRESHOLDS, + type IncidentSample +} from './incident-monitor.js' import type { AdmissionSelector } from './incident-selector.js' const directories: string[] = [] @@ -313,7 +316,7 @@ describe('relay incident live preflight', () => { it('retries freshness-only failures when explicitly requested', async () => { const stale = sample() stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const missing = sample() delete missing.sources['relay-logs'] const collect = vi.fn() @@ -334,7 +337,7 @@ describe('relay incident live preflight', () => { it('retries a first-wave stale sample and passes on the fresh one', async () => { const stale = sample() stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample()) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( @@ -348,7 +351,7 @@ describe('relay incident live preflight', () => { it('stops retrying when the next wait would exceed the evidence-age bound', async () => { const completedAt = now - 290_000 const stale = sample() - stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => stale) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( @@ -368,7 +371,7 @@ describe('relay incident live preflight', () => { const unhealthy = sample() unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => unhealthy) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( @@ -381,7 +384,7 @@ describe('relay incident live preflight', () => { it('fails closed after the bounded freshness retry window', async () => { const stale = sample() - stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => stale) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts index fc2c3a99751..2fcce3ed85d 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts @@ -7,6 +7,7 @@ import { suppliedIdentityToken } from './incident-monitor-cli.js' import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js' import { evaluateIncidentSample, + FRESHNESS_FAILURE_CODES, preDrainDryRunPassed, type IncidentSample } from './incident-monitor.js' @@ -18,12 +19,6 @@ const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000 // Matches the same-cap cell job timeout-minutes; bounds each predecessor wave. const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000 const WAVE_INDEX_PATTERN = /^[0-3]$/ -const FRESHNESS_FAILURE_CODES = new Set([ - 'signal_missing', - 'signal_stale', - 'source_missing', - 'source_stale' -]) export function livePreflightGcloud( gcloud: ReturnType, diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.ts index e090be7ea58..adfe6cad480 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-cli.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.ts @@ -50,6 +50,8 @@ const StateSchema = z.object({ continuityEvents: z.array(z.object({ recordedAt: z.string(), windowSequence: z.number().int().nonnegative(), + // Pre-2026-09-05 state files predate tolerated freshness gaps. + tolerated: z.boolean().default(false), failures: z.array(z.object({ code: z.string(), source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']), diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts index 09b7b16fa45..74054b6c0ba 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts @@ -93,7 +93,7 @@ describe('incident monitor sources', () => { }) it('zero-fills an expired sparse lock-wait point', async () => { - let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ points: [{ @@ -141,7 +141,7 @@ describe('incident monitor sources', () => { it('freshens a sparse zero without masking a recent nonzero lock wait', async () => { let value = 0 - const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const readAt = now + 11_879 const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.ts index 0b78c2c4f6b..a97bfe3df43 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.ts @@ -95,7 +95,7 @@ export const GOOGLE_METRICS: GoogleMetricDefinition[] = [ 'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"', aggregation: 'latest-max', emptyIsZero: true, - zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs }, { signal: 'cloud_sql.deadlocks', diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 4e1da9fab26..076cff3de3b 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import { evaluateIncidentSample, INCIDENT_CHECKPOINT_MINUTES, + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES, INCIDENT_MONITOR_THRESHOLDS, INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS, initialIncidentMonitorState, @@ -182,12 +183,49 @@ describe('incident monitor evaluator', () => { code: 'source_missing', source: 'relay-logs' }) - const stale = healthySample(startedAt - 180_001) + const stale = healthySample( + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) const failures = evaluateIncidentSample(stale, startedAt).failures expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true) expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true) }) + // Why: production run 33944873727 at 2026-09-05T04:46:09Z read + // cloud_sql.lock_waits 189 286 ms old and restarted a 15-minute window on + // Google's publish lag. Cloud SQL documents 60 s sampling plus up to 165 s of + // invisibility, so that age is Google's clock, not our fleet. + it('reads a 189-second cloud signal as fresh and holds the other sources at 180 s', () => { + const lagged = healthySample() + lagged.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, startedAt - 189_286) + expect(evaluateIncidentSample(lagged, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + const laggedDirector = healthySample() + laggedDirector.sources['director-admin']!.observedAt = + new Date(startedAt - 189_286).toISOString() + expect(evaluateIncidentSample(laggedDirector, startedAt).failures).toContainEqual( + expect.objectContaining({ code: 'source_stale', source: 'director-admin' }) + ) + }) + + it('still fails a cloud signal past the documented publish lag', () => { + const dark = healthySample() + dark.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal( + 0, + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + expect(evaluateIncidentSample(dark, startedAt).failures).toContainEqual( + expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + }) + ) + }) + it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => { const sample = healthySample() sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81) @@ -592,7 +630,7 @@ describe('incident monitor lifecycle', () => { 'restarts a %i-minute continuous window after stale telemetry', async (durationMinutes) => { let now = startedAt - let staleInjected = false + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const checkpoints: Array<[number, number]> = [] const state = initialIncidentMonitorState({ incidentId: 'incident-1', @@ -612,9 +650,11 @@ describe('incident monitor lifecycle', () => { now += ms }, collect: async () => { - if (!staleInjected && now === startedAt + 5 * 60_000) { - staleInjected = true - return healthySample(now - 180_001) + if (staleSamples > 0 && now >= startedAt + 5 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) } return healthySample(now) }, @@ -623,16 +663,20 @@ describe('incident monitor lifecycle', () => { checkpoints.push([summary.windowSequence, summary.checkpointMinute]) } }) + const restartMinute = 5 + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 expect(result.windowSequence).toBe(1) expect(result.windowStartedAt).toBe( - new Date(startedAt + 6 * 60_000).toISOString() + new Date(startedAt + restartMinute * 60_000).toISOString() ) expect(result.completedAt).toBe( - new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString() + new Date(startedAt + (durationMinutes + restartMinute) * 60_000).toISOString() ) expect(result.sampleCount).toBe(durationMinutes + 1) - expect(result.continuityEvents).toHaveLength(1) - expect(result.continuityEvents[0]!.failures).toEqual( + expect(result.continuityEvents.map((event) => event.tolerated)).toEqual([ + ...Array(INCIDENT_FRESHNESS_TOLERANCE_SAMPLES).fill(true), + false + ]) + expect(result.continuityEvents.at(-1)!.failures).toEqual( expect.arrayContaining([ expect.objectContaining({ code: 'source_stale' }) ]) @@ -642,6 +686,188 @@ describe('incident monitor lifecycle', () => { } ) + // Why: run 33944873727 on 2026-09-05 restarted at 04:46:09Z on a single + // 189-second cloud reading and then blew the 25-minute lineage cap, so a + // green fleet produced no verdict at all. One unread sample now continues the + // window; the sample is still checked against every threshold it can read. + it('carries a 15-minute window through a single stale cloud sample', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 10 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.windowStartedAt).toBe(new Date(startedAt).toISOString()) + expect(result.completedAt).toBe(new Date(startedAt + 15 * 60_000).toISOString()) + expect(result.sampleCount).toBe(16) + expect(result.frozenAt).toBeNull() + expect(result.continuityEvents).toEqual([{ + recordedAt: new Date(startedAt + 10 * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + })] + }]) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('gives a signal a fresh budget only after it reads fresh again', async () => { + let now = startedAt + const staleMinutes = new Set([3, 5, 6, 9, 10]) + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (staleMinutes.has((now - startedAt) / 60_000)) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.continuityEvents).toHaveLength(staleMinutes.size) + expect(result.continuityEvents.every((event) => event.tolerated)).toBe(true) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('does not hand a resumed monitor a fresh tolerance budget', async () => { + let now = startedAt + 3 * 60_000 + const resumed = { + ...initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }), + windowStartedAt: new Date(startedAt).toISOString(), + lastSampleAt: new Date(startedAt + 2 * 60_000).toISOString(), + sampleCount: 3, + totalSampleCount: 3, + continuityEvents: Array.from( + { length: INCIDENT_FRESHNESS_TOLERANCE_SAMPLES }, + (_, index) => ({ + recordedAt: new Date(startedAt + (index + 1) * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [{ + code: 'signal_stale', + source: 'cloud-monitoring' as const, + signal: 'cloud_sql.lock_waits' + }] + }) + ) + } + const stop = new Error('stop after the resumed sample') + await expect(runIncidentMonitor(resumed, { + now: () => now, + wait: async () => { + throw stop + }, + collect: async () => { + const sample = healthySample(now) + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + return sample + }, + persist: async (state) => { + expect(state.windowSequence).toBe(1) + expect(state.windowStartedAt).toBeNull() + expect(state.continuityEvents.at(-1)!.tolerated).toBe(false) + }, + checkpoint: async () => {} + })).rejects.toThrow(stop) + }) + + it('freezes on a threshold breach that arrives with a tolerated stale signal', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 2 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81, now) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.frozenAt).toBe(new Date(startedAt + 2 * 60_000).toISOString()) + expect(result.failures).toContainEqual(expect.objectContaining({ + code: 'threshold_max', + signal: 'cloud_sql.cpu' + })) + expect(preDrainDryRunPassed(result)).toBe(false) + }) + it('resets at the next fresh sample after a runner gap', async () => { let now = startedAt + 10 * 60_000 const state = { @@ -690,13 +916,21 @@ describe('incident monitor lifecycle', () => { durationMinutes: 15, intervalMs: 60_000 }) + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const result = await runIncidentMonitor(state, { now: () => now, wait: async (ms) => { now += ms }, - collect: async () => - healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now), + collect: async () => { + if (staleSamples > 0 && now >= startedAt + 10 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + } + return healthySample(now) + }, persist: async () => {}, checkpoint: async () => {} }) @@ -706,7 +940,7 @@ describe('incident monitor lifecycle', () => { ) expect(result.frozenAt).not.toBeNull() expect(result.windowSequence).toBe(1) - expect(result.sampleCount).toBe(15) + expect(result.sampleCount).toBe(13) expect(result.failures).toContainEqual({ code: 'continuity_deadline_exceeded', source: 'active-probe', diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index a121568d918..2785eb573af 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -6,7 +6,27 @@ import { export const INCIDENT_MONITOR_THRESHOLDS = { activeProbeMaxAgeMs: 60_000, - cloudDataMaxAgeMs: 180_000, + // Why: Cloud Monitoring publishes on Google's clock, not ours. Per the metric + // list read 2026-09-05, Cloud Run instance_count / cpu / memory / + // max_request_concurrencies / request_count are "Sampled every 60 seconds. + // After sampling, data is not visible for up to 120 seconds" (60+120=180 s), + // and Cloud SQL cpu / memory / num_backends / backends_in_wait / + // deadlock_count say "up to 165 seconds" (60+165=225 s). Window-sum signals + // age differently: observedAt is the newest point in the 5-minute query + // window, so a label series that stops emitting reads as 300 s old while its + // summed value is still complete. 330 s clears the worst of the three (the + // 300 s query window) plus ~30 s of collect-to-evaluate latency. The old + // 180 s bar restarted healthy 15-minute windows at 181 s, 189 s and 255 s on + // 2026-09-04/05, once burning the whole 25-minute lineage with no verdict. + cloudDataMaxAgeMs: 330_000, + // Why: the director admin API answers live on our own request, so hold its + // freshness bar where it sat while it shared cloudDataMaxAgeMs. + directorAdminMaxAgeMs: 180_000, + // Why: how long a nonzero backends-in-wait point is carried before it reads as + // zero. Held at the pre-2026-09-05 cloud bar: carrying it for the full + // cloudDataMaxAgeMs would hand the evaluator a point older than its own + // freshness bar as soon as collection latency is added. + cloudLockWaitCarryMs: 180_000, relayLogMaxAgeMs: 180_000, heartbeatMaxAgeMs: 45_000, endpointLatencyMs: 2_000, @@ -175,6 +195,7 @@ export type IncidentMonitorState = { continuityEvents: { recordedAt: string windowSequence: number + tolerated: boolean failures: IncidentFailure[] }[] frozenAt: string | null @@ -307,7 +328,7 @@ const SOURCE_MAX_AGE: Record = { 'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs, 'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs, 'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs, - 'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 'director-admin': INCIDENT_MONITOR_THRESHOLDS.directorAdminMaxAgeMs } function ageMs(timestamp: string, nowMs: number): number { @@ -608,14 +629,59 @@ function checkpointMinutes(durationMinutes: number): number[] { return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes) } -const CONTINUITY_FAILURE_CODES = new Set([ - 'collector_failed', - 'monitor_gap', +// Freshness-only failures: we could not read a signal this sample. Distinct from +// collector_failed / monitor_gap, where the whole sample is absent. +export const FRESHNESS_FAILURE_CODES = new Set([ + 'signal_missing', 'signal_stale', 'source_missing', 'source_stale' ]) +const CONTINUITY_FAILURE_CODES = new Set([ + 'collector_failed', + 'monitor_gap', + ...FRESHNESS_FAILURE_CODES +]) + +// Why: Cloud Monitoring overshoots its own publish bar, and one unread sample is +// not evidence of an unhealthy fleet. Under the 25-minute lineage cap a restart +// past minute 10 costs the entire verdict, so a healthy fleet produced none on +// 2026-09-05. A signal may miss this many consecutive samples before the window +// restarts; the sample is still evaluated against every threshold it can read, +// and a threshold breach still freezes the run outright. +export const INCIDENT_FRESHNESS_TOLERANCE_SAMPLES = 2 + +function freshnessKey(failure: IncidentFailure): string { + return `${failure.source}/${failure.signal ?? '*'}` +} + +// Rebuild the per-signal tolerated streak from the trailing continuity events so a +// resumed monitor cannot hand a signal a fresh budget. +function resumeFreshnessStreaks( + state: IncidentMonitorState +): Map { + const events = state.continuityEvents + const streaks = new Map() + const last = events[events.length - 1] + if (!last?.tolerated) return streaks + for (const key of new Set(last.failures.map(freshnessKey))) { + let streak = 0 + let laterAt: number | null = null + for (let index = events.length - 1; index >= 0; index--) { + const event = events[index]! + const recordedAt = Date.parse(event.recordedAt) + if (!event.tolerated) break + if (laterAt !== null && laterAt - recordedAt > state.intervalMs * 1.5) break + if (!event.failures.some((failure) => freshnessKey(failure) === key)) break + streak++ + laterAt = recordedAt + } + streaks.set(key, streak) + } + return streaks +} + function resetContinuousWindow( state: IncidentMonitorState, recordedAt: string, @@ -631,6 +697,7 @@ function resetContinuousWindow( state.continuityEvents.push({ recordedAt, windowSequence: state.windowSequence, + tolerated: false, failures }) } @@ -681,6 +748,7 @@ export async function runIncidentMonitor( await dependencies.persist(state) return state } + const freshnessStreaks = resumeFreshnessStreaks(state) while (state.completedAt === null) { if (dependencies.now() > lineageDeadlineMs) { completeContinuityDeadline(state, dependencies.now(), lineageStartMs) @@ -715,9 +783,34 @@ export async function runIncidentMonitor( const thresholdFailures = evaluation.failures.filter((failure) => !CONTINUITY_FAILURE_CODES.has(failure.code) ) - if (continuityFailures.length > 0) { + const toleratedKeys = new Set( + state.windowStartedAt !== null && + continuityFailures.length > 0 && + continuityFailures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code)) + ? continuityFailures.map(freshnessKey) + : [] + ) + for (const key of [...freshnessStreaks.keys()]) { + if (!toleratedKeys.has(key)) freshnessStreaks.delete(key) + } + let tolerated = toleratedKeys.size > 0 + for (const key of toleratedKeys) { + const streak = (freshnessStreaks.get(key) ?? 0) + 1 + freshnessStreaks.set(key, streak) + if (streak > INCIDENT_FRESHNESS_TOLERANCE_SAMPLES) tolerated = false + } + if (continuityFailures.length > 0 && !tolerated) { + freshnessStreaks.clear() resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures) } else { + if (tolerated) { + state.continuityEvents.push({ + recordedAt: evaluation.evaluatedAt, + windowSequence: state.windowSequence, + tolerated: true, + failures: continuityFailures + }) + } if (state.windowStartedAt === null) { state.windowStartedAt = evaluation.evaluatedAt } diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 870c95dd413..8a8dfda1495 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -73,6 +73,13 @@ for a committed forward-recovery gate. Durable files default to gap resets the active window at the next fresh sample and preserves the prior window evidence. A threshold freeze never clears automatically. +A signal that reads missing or stale may miss up to two consecutive samples +without restarting the window. The sample still counts and is still checked +against every threshold it can read, and each tolerated gap is recorded in +`continuityEvents` with `tolerated: true`. A third consecutive miss of the same +signal, a failed collector, a runner gap, or any threshold breach restarts or +freezes as before. + A production candidate or multi-target mutation must download the exact dry-run artifact by workflow run ID and attempt. It verifies the artifact hashes and provenance, requires a green completed 15-minute state no older @@ -89,7 +96,8 @@ durably marked consumed before mutation and cannot authorize another run. | Signal | Freeze condition | | --- | ---: | | Active probe age | over 60 seconds | -| Cloud/log data age | over 180 seconds | +| Cloud Monitoring data age | over 330 seconds | +| Relay log and director admin data age | over 180 seconds | | Cell heartbeat age | over 45 seconds | | Endpoint latency | over 2,000 ms | | Cloud SQL CPU | over 80% | @@ -155,6 +163,25 @@ heartbeats, and matching live admission. separate it from today's baseline; the exhausted-retry bar (incident peak 467 vs bar 300), director concurrency, and the pool bars carry that role. Re-tighten after the fleet is on the 500 ms lock wait. +- Raised the Cloud Monitoring freshness bar from 180 s to 330 s and let a + freshness-only failure miss up to two consecutive samples without restarting + the window (2026-09-05). Basis: Google's metric list documents Cloud Run + `request_count`, `container/instance_count`, `container/cpu/utilizations`, + `container/memory/utilizations` and `container/max_request_concurrencies` as + "Sampled every 60 seconds. After sampling, data is not visible for up to 120 + seconds", and Cloud SQL `database/cpu/utilization`, + `database/memory/utilization`, `database/postgresql/num_backends`, + `database/postgresql/backends_in_wait` and `database/postgresql/deadlock_count` + as "up to 165 seconds", so the newest visible point is up to 180 s and 225 s + old respectively. Window-sum signals age further: `observedAt` is the newest + point in the 5-minute query window, so a label series that stops emitting + reads as 300 s old while its summed value is complete. The old bar sat under + all three. Production on 2026-09-04/05 restarted healthy 15-minute windows at + 181 s and 255 s (`auth.errors`, run 33928912676) and at 189 s + (`cloud_sql.lock_waits`, run 33944873727), and the last of those then blew the + 25-minute lineage cap at 1 500 004 ms, so a green fleet produced no verdict. + The director admin bar stays at 180 s and the nonzero lock-wait carry window + stays at 180 s; both publish on our own cadence. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters From 41b520259ec8f5c3f176b95ccdc172b6a8be310c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:24:18 -0700 Subject: [PATCH 07/26] ci(package): retry apt fetches and docker builds behind the Ubuntu mirror (#18797) The package job builds three Docker images whose apt-get update/install hit archive.ubuntu.com with no retry, timeout, or mirror fallback. When the mirror is mid-sync every build dies in one of three ways: - per-package fetch stalls (~64 s each, `Ign:` lines) until the runner's 10-minute docker build timeout fires: https://github.com/stablyai/orca/actions/runs/33935104546/job/101221447425 - `apt-get update` exit 100 with `Hash Sum mismatch` on noble-updates/restricted/Packages.gz: https://github.com/stablyai/orca/actions/runs/33935104546/job/101226099525 - `apt-get update` exit 100 with `File has unexpected size ... Mirror sync in progress?`: https://github.com/stablyai/orca/actions/runs/33935244026/job/101231083497 Each Dockerfile now retries `apt-get update` up to five times with Acquire::Retries and a 30 s HTTP timeout, clearing /var/lib/apt/lists between attempts so a half-synced index is never reused, and passes the same acquire options to `apt-get install`. Each runner script retries the whole `docker build` once when the first attempt fails or times out. --- config/docker/cli-launch-contract/Dockerfile | 9 +++-- config/docker/headless-pairing/Dockerfile | 9 +++-- .../docker/headless-serve-shutdown/Dockerfile | 9 +++-- .../run-headless-linux-pairing-docker.mjs | 13 +++++-- .../run-headless-serve-shutdown-docker.mjs | 12 +++++-- .../run-linux-cli-launch-contract-docker.mjs | 34 +++++++++++-------- 6 files changed, 62 insertions(+), 24 deletions(-) diff --git a/config/docker/cli-launch-contract/Dockerfile b/config/docker/cli-launch-contract/Dockerfile index f6a618a8ece..c90cbcd979c 100644 --- a/config/docker/cli-launch-contract/Dockerfile +++ b/config/docker/cli-launch-contract/Dockerfile @@ -6,8 +6,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive # Install Electron's link-time libraries without adding a display server or FUSE. -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ coreutils \ diff --git a/config/docker/headless-pairing/Dockerfile b/config/docker/headless-pairing/Dockerfile index 03664f68b0d..e4b4cafeefc 100644 --- a/config/docker/headless-pairing/Dockerfile +++ b/config/docker/headless-pairing/Dockerfile @@ -5,8 +5,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/docker/headless-serve-shutdown/Dockerfile b/config/docker/headless-serve-shutdown/Dockerfile index 13b1ed2b69f..8ee7b942499 100644 --- a/config/docker/headless-serve-shutdown/Dockerfile +++ b/config/docker/headless-serve-shutdown/Dockerfile @@ -2,8 +2,13 @@ FROM ubuntu@sha256:678c6550cc43645e08669028bc177f50be4e7c5b8cca677067b1914d4afc7 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/scripts/run-headless-linux-pairing-docker.mjs b/config/scripts/run-headless-linux-pairing-docker.mjs index 635c66348cc..799b8d73ab1 100644 --- a/config/scripts/run-headless-linux-pairing-docker.mjs +++ b/config/scripts/run-headless-linux-pairing-docker.mjs @@ -70,7 +70,7 @@ function valueAfter(flag) { function buildImage(image) { console.log(`Building ${image.name} fixture...`) - docker([ + const buildArgs = [ 'build', '--build-arg', `BASE_IMAGE=${image.base}`, @@ -81,7 +81,16 @@ function buildImage(image) { '-t', image.tag, '.' - ]) + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + try { + docker(buildArgs) + } catch (error) { + console.error( + `${error instanceof Error ? error.message : String(error)}\nRetrying docker build once...` + ) + docker(buildArgs) + } } function extractAppImage(image) { diff --git a/config/scripts/run-headless-serve-shutdown-docker.mjs b/config/scripts/run-headless-serve-shutdown-docker.mjs index 184713c41a0..d8dcdd345ad 100755 --- a/config/scripts/run-headless-serve-shutdown-docker.mjs +++ b/config/scripts/run-headless-serve-shutdown-docker.mjs @@ -43,7 +43,7 @@ const artifactVolume = `orca-headless-serve-shutdown-${suffix}` const sha256 = createHash('sha256').update(readFileSync(appImage)).digest('hex') try { - docker([ + const buildArgs = [ 'build', '--platform', platform, @@ -52,7 +52,15 @@ try { '-t', image, shutdownDockerDirectory - ]) + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + const firstBuild = docker(buildArgs, { allowFailure: true }) + if (firstBuild.status !== 0) { + process.stderr.write( + `${firstBuild.stdout}${firstBuild.stderr}\ndocker build failed with status ${firstBuild.status}; retrying once...\n` + ) + docker(buildArgs) + } docker(['volume', 'create', artifactVolume]) runDesktopStartupOracle({ image, appImage, platform }) docker([ diff --git a/config/scripts/run-linux-cli-launch-contract-docker.mjs b/config/scripts/run-linux-cli-launch-contract-docker.mjs index 901e0877e85..bd414947026 100755 --- a/config/scripts/run-linux-cli-launch-contract-docker.mjs +++ b/config/scripts/run-linux-cli-launch-contract-docker.mjs @@ -173,20 +173,26 @@ function runCase(caseName) { function buildImage() { console.log(`Building ${tag}…`) - docker( - [ - 'build', - ...dockerPlatformArgs, - '--build-arg', - `BASE_IMAGE=${base}`, - '-f', - 'config/docker/cli-launch-contract/Dockerfile', - '-t', - tag, - 'config/docker/cli-launch-contract' - ], - { timeoutMs: BUILD_TIMEOUT_MS } - ) + const buildArgs = [ + 'build', + ...dockerPlatformArgs, + '--build-arg', + `BASE_IMAGE=${base}`, + '-f', + 'config/docker/cli-launch-contract/Dockerfile', + '-t', + tag, + 'config/docker/cli-launch-contract' + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + try { + docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS }) + } catch (error) { + console.error( + `${error instanceof Error ? error.message : String(error)}\nRetrying docker build once…` + ) + docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS }) + } } // Extract unprivileged so chrome-sandbox is not root-owned setuid. From b33d1972bce27f8d04c25087dee05f82b9740ee0 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:40:38 -0700 Subject: [PATCH 08/26] docs(relay): correct why the ConPTY teardown asset diverges from the desktop patch (#18636) The divergence pinned by #18601 is real and worth keeping, but its stated reason was wrong. It claimed the desktop patch carries the early conin placement "and therefore the +2 File / +1 Process regression, measured against its exact installed tree" -- i.e. that the shipped desktop app leaks because of its own leak fix. It does not, for any terminal a user opens. node-pty defaults `_useConptyDll` to false. Every desktop site that opens a pane sets it true (`local-pty-utils.ts` twice, `native-pty-spawn.ts`), as does the `windows-conpty-warmup.ts` warm-up, so they take the `else` branch, where upstream already destroys the input socket. The relay passes no such option (`src/relay/pty-handler.ts`) and takes the `!useConptyDll` branch -- the one both this asset and the desktop patch edit. The desktop is not entirely off that branch, though: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and `codex-pty-rate-limit-probe.ts` omit the option, recur, and tear down through `kill()`, so the hunk is live there -- just never for a visible pane. Whether the early placement costs the same +2 File / +1 Process across a probe's lifecycle is unmeasured; the numbers in this comment were taken on relay-style spawn/kill cycles, and the comment now says so. What is settled is the replaced claim: not every Windows user, and not every terminal. Those two probes were missed three enumerations running because they use `await import('node-pty')`, which no static-import grep finds. The comment now tells the next reader to grep for `node-pty` instead. The measurement that produced the wrong claim was taken by a standalone harness that passed no `useConptyDll` and so defaulted into the branch it was not trying to measure -- the same standalone-is-not-the-real-host trap #18601's own body warns about, one level down. Also refreshes the self-exit paragraph, which #18635 made stale. That leak is now fixed for the desktop, and the note records why the fix cannot reach a Windows relay. The fix is mostly native (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`, and all three delivery paths stop short of Windows: pnpm patches do not cross the SSH boundary; `MATRIX_SLOTS` in `build-orcad-prebuilds.mjs` has no win32 entry; and the one relay asset that does patch native source and rebuild on the host (`node-pty-1.1.0-master-cloexec-patch.cjs`) returns `skipped:unsupported-platform` for anything but linux/darwin. #18635's flat self-exit relay numbers were measured against a locally rebuilt binary, so they describe the relay code path on a patched tree, not the tree a relay host installs -- the note says so explicitly rather than leaving the next reader to conflate them. Assertion and hashes unchanged: the relay must still release conin after the console-list fork, and a patch sync must still not copy the early placement onto the relay's branch, where it does cost +2 File and +1 Process per terminal. Only the justification changes, plus the test name, which said "like the desktop patch" where it meant "unlike the desktop patch placement". --- ...e-pty-1.1.0-windows-pty-teardown-patch.cjs | 81 +++++++++++++++---- ...de-pty-windows-pty-teardown-patch.test.mjs | 32 +++++--- 2 files changed, 85 insertions(+), 28 deletions(-) diff --git a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs index dd26784ee46..1e908754dc6 100644 --- a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs +++ b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs @@ -12,14 +12,15 @@ const { join, resolve } = require('node:path') * `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`. * Every terminal leaks one File handle for the life of the host process. * - * The obvious fix -- and the one the desktop patch ships -- releases it at the TOP of the branch, - * before `_getConsoleProcessList()` forks and before the native kill. That is measurably worse than - * leaving the leak alone: teardown aborts partway, the forked console-list agent is never reaped, - * and both pipe handles stay alive instead of one. This asset releases it at the END of the branch - * instead, after the fork and the kill have already happened. + * The obvious fix -- and the placement `config/patches/node-pty@1.1.0.patch` uses -- releases it at + * the TOP of the branch, before `_getConsoleProcessList()` forks and before the native kill. That is + * measurably worse than leaving the leak alone: teardown aborts partway, the forked console-list + * agent is never reaped, and both pipe handles stay alive instead of one. This asset releases it at + * the END of the branch instead, after the fork and the kill have already happened. * * Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type - * (identical numbers standalone and through a real relay): + * (identical numbers standalone and through a real relay). Every row is the NON-DLL branch, which + * is the branch a relay runs -- see the divergence note below for why that matters: * * published node-pty File +1/terminal, Process flat * desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE @@ -34,18 +35,64 @@ const { join, resolve } = require('node:path') * Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm * patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there. * - * DELIBERATE DIVERGENCE FROM THE DESKTOP: the desktop patch has the early placement and therefore - * the +2 File / +1 Process regression, measured against its exact installed tree. Correcting it - * there is a separate change with its own verification, so the two trees differ on this one hunk on - * purpose, and the test pins that so a future "sync the patches" does not copy the bug back. + * DELIBERATE DIVERGENCE FROM THE DESKTOP, AND WHY IT IS NOT A DESKTOP-TERMINAL BUG: the two hosts + * do not run the same branch of `kill()`. node-pty defaults `_useConptyDll` to false + * (`windowsPtyAgent.js`). Every desktop site that opens a terminal pane sets it true -- + * `local-pty-utils.ts` (two) and `native-pty-spawn.ts` -- as does the `windows-conpty-warmup.ts` + * warm-up, so all of those take the `else` branch, where UPSTREAM ALREADY destroys the input + * socket. The relay passes no such option (`src/relay/pty-handler.ts`), so it takes the + * `!useConptyDll` branch -- the one this asset and the desktop patch both edit. * - * NOT ADDRESSED, AND A SEPARATE DEFECT THAT IS STILL OPEN: a terminal that exits on its own is - * still torn down through `kill()` -- both hosts call `destroy()` on natural exit and - * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, and the - * ordering this patch relies on does not hold. Measured over 20 self-exit cycles with that - * `destroy()` issued: published +3 File/+1 Process per terminal, desktop-patched +2/+1, this tree - * +2/+1. So this patch does not close it and the desktop patch does not either. It is reachable - * for every Windows user, local and relay, on every terminal closed by typing `exit`. + * THE DESKTOP IS NOT ENTIRELY OFF THAT BRANCH. Two desktop sites omit the option and so run it + * too: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and + * `codex-pty-rate-limit-probe.ts`. Both recur -- their fetchers poll -- and both tear down through + * `kill()`, so this hunk is live on the desktop, just never for a pane a user can see. Do not + * restate this as "the desktop never executes that branch": that sentence stood here for two + * revisions and is false. + * + * What the numbers above therefore do NOT cover: they were measured on relay-style spawn/kill + * cycles. Whether the early placement costs the same +2 File / +1 Process across a probe's + * lifecycle is UNMEASURED -- plausible, not established, and worth measuring before anyone quotes + * a desktop figure. What IS settled is the claim this comment replaced: that the desktop patch made + * every Windows user worse off ON EVERY TERMINAL. Terminals take the DLL branch, and the harness + * that produced that claim defaulted into the branch it was not trying to measure. + * + * The divergence is therefore about which branch each host runs for the workload that matters, not + * about a regression in the terminals users open. The test still pins it, because a future "sync + * the patches" would put the early placement onto the relay's branch, where it does cost +2 File + * and +1 Process per terminal. + * + * If you extend this enumeration, grep for `node-pty` rather than for a static import: those two + * probes were missed three times because they use `await import('node-pty')`. + * + * THE SELF-EXIT LEAK: FIXED FOR THE DESKTOP BY #18635, STILL LIVE ON A RELAY. A terminal that exits + * on its own is also torn down through `kill()` -- both hosts call `destroy()` on natural exit and + * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, so the ordering + * this asset relies on does not hold. Measured over 20 self-exit cycles on the NON-DLL branch: + * published +3 File/+1 Process per terminal, desktop patch placement +2/+1, this tree +2/+1. This + * asset does not close it. + * + * #18635 does, in `config/patches/node-pty@1.1.0.patch`: the baton outlives the shell so `PtyKill` + * still reaches `ClosePseudoConsole`, plus an unconditional conout dispose on the DLL branch. That + * fix does not reach a Windows relay, and no hunk in THIS file can carry it, because it is mostly + * NATIVE (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`. Three delivery paths exist + * and none currently covers Windows: + * + * - the pnpm patch does not cross the SSH boundary -- the remote `npm install` yields upstream's + * unpatched node-pty; + * - the orcad prebuild matrix has no win32 entry (`MATRIX_SLOTS`, + * `config/scripts/build-orcad-prebuilds.mjs`), so no Windows binary is ever compiled from + * patched source to ship; + * - a relay asset CAN patch native source and rebuild on the host -- that is exactly what + * `node-pty-1.1.0-master-cloexec-patch.cjs` does -- but it returns + * `skipped:unsupported-platform` for anything but linux/darwin. Extending it to win32 means + * requiring an MSVC toolchain on the relay host, a far heavier precondition than on Linux, + * where node-gyp already runs at install time. + * + * So a Windows SSH relay still leaks a pseudoconsole per self-exiting terminal, and closing it is a + * DELIVERY problem, not another hunk here. Do not read #18635's flat self-exit relay numbers as + * covering deployed relays: they were measured against a locally rebuilt binary, so they describe + * the relay CODE PATH on a patched tree, not the tree a relay host actually installs. */ const EXPECTED_NODE_PTY_VERSION = '1.1.0' diff --git a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs index 64fb1b056b8..ba3e64bfa2f 100644 --- a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs +++ b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs @@ -1,11 +1,18 @@ -// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with the -// desktop's own node-pty patch. pnpm patches do not cross the SSH boundary, so a relay runs the tree -// `npm install` put there; the desktop had this fix and the relay did not, and every terminal on a -// Windows SSH host leaked one File handle for the life of the relay process. +// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with +// `config/patches/node-pty@1.1.0.patch`. pnpm patches do not cross the SSH boundary, so a relay runs +// the tree `npm install` put there, and every terminal on a Windows SSH host leaked one File handle +// for the life of the relay process. // -// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- what the -// desktop patch does -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a new -// Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- the placement +// the desktop patch uses -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a +// new Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// +// Those numbers are the `!useConptyDll` branch, which is the branch a RELAY runs. Every desktop +// site that opens a terminal pane sets `useConptyDll: true` and takes the other branch, where +// upstream already destroys the input socket. Two hidden rate-limit probes +// (`src/main/rate-limits/claude-pty.ts`, `codex-pty-rate-limit-probe.ts`) do omit the option and so +// do run this hunk, but no user-visible pane does. The divergence pinned below is about which +// branch each host runs for terminals -- not about a regression in the panes users open. import { createRequire } from 'node:module' import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { join, resolve } from 'node:path' @@ -90,9 +97,11 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => { ) }) - // The one hunk that must NOT match the desktop, and the reason is measured, not stylistic: - // releasing conin before `_getConsoleProcessList()` forks aborts teardown partway. - it('releases conin after the console-list fork, not before it like the desktop patch', () => { + // The one hunk that must NOT match the desktop patch, and the reason is measured, not stylistic: + // on the branch a relay runs, releasing conin before `_getConsoleProcessList()` forks aborts + // teardown partway. Desktop terminal panes take the other branch, so no pane is affected either + // way; what this guards is a patch sync putting the early placement onto the relay's branch. + it('releases conin after the console-list fork, unlike the desktop patch placement', () => { const fixture = writeNodePtyFixture('1.1.0') patchNodePtyWindowsTeardown(fixture.root) const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8') @@ -108,7 +117,8 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => { expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( branch.indexOf('this._getConsoleProcessList()') ) - // Pinned so a future "sync the relay asset to config/patches" cannot copy the regression back. + // Pinned so a future "sync the relay asset to config/patches" cannot copy the early placement + // onto the relay's branch, where it costs +2 File and +1 Process per terminal. expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8')) }) From 3f84c358f0fc1a5468a4a049490f75fb8408f264 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:27:30 -0700 Subject: [PATCH 09/26] fix: recover from an alternate shell install, and close the relay duplicate-echo gap (#18796) * fix: recover from an alternate shell install, and close the relay echo gap #18768: a startup profile that `exec`s a second install of the same shell keeps the pid but loses the wrapper's ready marker, and the recovery probe rejected the replacement's different canonical path -- costing plain Codex the full 15s barrier. The probe now also accepts an install that the pane's own PATH resolves, so a binary merely named bash/zsh outside it stays rejected. #18767: the SSH relay left plain Codex on early startup delivery, displaying the launch twice under a slow profile. The shell is the host's to know, so the relay now folds it into the same rule the daemon uses, and the SSH background client waits for the marker on any Codex launch. Bracketed paste is now gated on an observed marker rather than on the intent to wait, so a fallback release on a host shell that never publishes one submits raw. The non-daemon local provider needs no change: it hands Codex to the wrapper's own prompt hook and never writes it into the PTY. * review: correct an overclaiming comment, announce a silent skip, drop a shim Readiness review of #18796. The delivery comment claimed waiting "costs nothing", which is only true on a host that arms the marker -- fish, sh, Windows and hosts predating #18767 release on the fallback instead. Say so. The alternate-install recovery tests skip on usrmerge hosts, where /usr/bin/bash resolves back to /bin/bash; announce that rather than reading as coverage that does not exist. Import the line-editor predicate from shared directly instead of through a re-export left on daemon/shell-ready. --- src/main/daemon/daemon-pty-session-spawn.ts | 7 +- .../pty-subprocess/shell-launch-plan.ts | 10 +- ...67-shell-ready-marker-lost-to-exec.test.ts | 68 +++++++++++- .../daemon/session-shell-ready-barrier.ts | 2 +- src/main/daemon/shell-ready.ts | 5 - .../local-pty-finalize-environment.ts | 3 + src/main/shell-prompt-readiness-probe.test.ts | 101 +++++++++++++++++- src/main/shell-prompt-readiness-probe.ts | 29 +++-- ...y-handler-startup-command-delivery.test.ts | 67 ++++++++++++ src/relay/pty-handler.ts | 6 +- ...ch-agent-background-session-remote.test.ts | 27 +++++ .../lib/launch-agent-background-session.ts | 12 +-- .../ssh-background-startup-delivery.test.ts | 43 ++++++++ .../lib/ssh-background-startup-delivery.ts | 39 ++++++- src/shared/codex-startup-delivery.test.ts | 44 +++++++- src/shared/codex-startup-delivery.ts | 17 ++- src/shared/shell-process-readiness.test.ts | 48 ++++++++- src/shared/shell-process-readiness.ts | 50 +++++++-- src/shared/shell-ready-marker-timing.ts | 18 ++++ 19 files changed, 540 insertions(+), 56 deletions(-) create mode 100644 src/shared/shell-ready-marker-timing.ts diff --git a/src/main/daemon/daemon-pty-session-spawn.ts b/src/main/daemon/daemon-pty-session-spawn.ts index 70519439919..bbb899b64a3 100644 --- a/src/main/daemon/daemon-pty-session-spawn.ts +++ b/src/main/daemon/daemon-pty-session-spawn.ts @@ -12,11 +12,8 @@ import { DaemonPtySpawnResult } from './daemon-pty-spawn-result' import type { DaemonPtySpawnContext } from './daemon-pty-spawn-request' import type { ColdRestoreInfo } from './history-reader' import { mintPtySessionId } from './pty-session-id' -import { - shellPathSupportsPtyStartupBarrier, - shellReadyMarkerComesFromLineEditor, - resolvePtyShellPath -} from './shell-ready' +import { shellPathSupportsPtyStartupBarrier, resolvePtyShellPath } from './shell-ready' +import { shellReadyMarkerComesFromLineEditor } from '../../shared/shell-ready-marker-timing' import { getRecoveredHistorySeedSegments } from './terminal-history-seed-segments' import { AGENT_SESSION_CLAIM_DAEMON_PROTOCOL_VERSION, type CreateOrAttachResult } from './types' import { normalizeWslColdRestoreCwd } from './wsl-cold-restore-cwd' diff --git a/src/main/daemon/pty-subprocess/shell-launch-plan.ts b/src/main/daemon/pty-subprocess/shell-launch-plan.ts index 12e049d9f72..ba60e592743 100644 --- a/src/main/daemon/pty-subprocess/shell-launch-plan.ts +++ b/src/main/daemon/pty-subprocess/shell-launch-plan.ts @@ -35,11 +35,7 @@ import { } from '../../../shared/agent-process-recognition' import { ORCA_HERMES_STARTUP_QUERY_ENV } from '../../../shared/hermes-startup-query' import { WINDOWS_GIT_BASH_SHELL } from '../../../shared/windows-terminal-shell' -import { - getShellLaunchConfig, - resolvePtyShellPath, - shellReadyMarkerComesFromLineEditor -} from '../shell-ready' +import { getShellLaunchConfig, resolvePtyShellPath } from '../shell-ready' import { resolveWslSessionContext } from '../wsl-session-context' import { finalizeDaemonPtyEnvironment, rescrubDaemonPtyEnvironment } from './spawn-environment' import type { PtySubprocessOptions } from '../pty-subprocess' @@ -196,10 +192,10 @@ export function createPtyShellLaunchPlan( const waitsForShellReady = Boolean(opts.command) && (startupAgentRecognition?.agent !== 'codex' || - shellReadyMarkerComesFromLineEditor(shellPath) || shouldUseShellReadyStartupDelivery({ command: opts.command, - startupCommandDelivery: opts.startupCommandDelivery + startupCommandDelivery: opts.startupCommandDelivery, + shellPath })) delete env.ORCA_SHELL_FEATURES const shellLaunch = getShellLaunchConfig( diff --git a/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts b/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts index 52462e3a2eb..a1db9461070 100644 --- a/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts +++ b/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts @@ -1,7 +1,7 @@ import { spawnSync } from 'node:child_process' -import { existsSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { existsSync, mkdtempSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join } from 'node:path' import { describe, expect, it, vi } from 'vitest' import { createPtySubprocess } from './pty-subprocess' import { Session } from './session' @@ -10,6 +10,20 @@ const describePosix = process.platform === 'win32' ? describe.skip : describe const hasZsh = process.platform !== 'win32' && spawnSync('/bin/zsh', ['--version']).status === 0 const hasBash = process.platform !== 'win32' && spawnSync('/bin/bash', ['--version']).status === 0 const COMMAND_OUTPUT = 'ORCA_STARTUP_COMMAND_RAN' +// A second Bash install with its own canonical path -- the shape a login profile +// switches to (`exec /opt/homebrew/bin/bash`) and the one #18768 stalled on. A +// symlink cannot stand in: both sides are realpath'd before they are compared. +const alternateBashPath = ['/opt/homebrew/bin/bash', '/usr/local/bin/bash', '/usr/bin/bash'].find( + (candidate) => + hasBash && existsSync(candidate) && realpathSync(candidate) !== realpathSync('/bin/bash') +) +if (process.platform !== 'win32' && !alternateBashPath) { + // Why announced: usrmerge hosts resolve /usr/bin/bash back to /bin/bash, so these + // two skip on most Linux CI. A silent skip reads as coverage that does not exist. + console.warn( + '[repro-13767] no second Bash install with a distinct realpath; skipping the alternate-install recovery tests' + ) +} const READ_STARTED_FILE = '.orca-read-started' type ShellFixture = { @@ -122,7 +136,8 @@ type RunningFixture = { async function startFixture( fixture: ShellFixture, startupContent: string, - extraFiles: Record = {} + extraFiles: Record = {}, + pathEnv: string = process.env.PATH ?? '/usr/bin:/bin' ): Promise { const tempHome = mkdtempSync(join(tmpdir(), 'orca-shell-ready-exec-')) const previousHome = process.env.HOME @@ -150,7 +165,7 @@ async function startFixture( shellOverride: fixture.shellPath, env: { HOME: tempHome, - PATH: process.env.PATH ?? '/usr/bin:/bin', + PATH: pathEnv, SHELL: fixture.shellPath, TERM: 'xterm-256color' }, @@ -416,4 +431,49 @@ fi }, 10_000 ) + + const bashFixture = FIXTURES[2] as ShellFixture + const alternateBashTest = alternateBashPath ? it : it.skip + const alternateBashProfile = `if [[ -z "\${ORCA_EXEC_REPRO_DONE:-}" ]]; then + export ORCA_EXEC_REPRO_DONE=1 + exec ${alternateBashPath ?? '/bin/bash'} --noprofile --norc -l -i +fi +` + + alternateBashTest( + 'releases at the prompt of a second Bash install the pane PATH resolves', + async () => { + const running = await startFixture( + bashFixture, + alternateBashProfile, + {}, + `${dirname(alternateBashPath ?? '/bin/bash')}:/usr/bin:/bin` + ) + try { + await waitForOutput(running.subscribe, () => running.output().includes(COMMAND_OUTPUT)) + expect(running.session.shellState).toBe('ready') + expect(count(running.output(), COMMAND_OUTPUT)).toBe(1) + expect(running.output()).not.toContain('orca-shell-start') + } finally { + await running.cleanup() + } + }, + 10_000 + ) + + alternateBashTest( + 'does not trust a Bash install that the pane PATH cannot reach', + async () => { + const running = await startFixture(bashFixture, alternateBashProfile, {}, '/usr/bin:/bin') + try { + await waitForOutput(running.subscribe, () => running.output().includes('$')) + await new Promise((resolve) => setTimeout(resolve, 500)) + expect(running.session.shellState).toBe('pending') + expect(running.output()).not.toContain(COMMAND_OUTPUT) + } finally { + await running.cleanup() + } + }, + 10_000 + ) }) diff --git a/src/main/daemon/session-shell-ready-barrier.ts b/src/main/daemon/session-shell-ready-barrier.ts index 538fd57a49a..6fce89af89e 100644 --- a/src/main/daemon/session-shell-ready-barrier.ts +++ b/src/main/daemon/session-shell-ready-barrier.ts @@ -1,4 +1,4 @@ -import { shellReadyMarkerComesFromLineEditor } from './shell-ready' +import { shellReadyMarkerComesFromLineEditor } from '../../shared/shell-ready-marker-timing' import { installDeviceAttributesResponder, STARTUP_DA1_RESPONSE diff --git a/src/main/daemon/shell-ready.ts b/src/main/daemon/shell-ready.ts index 47acc567f6f..5208f9ced6b 100644 --- a/src/main/daemon/shell-ready.ts +++ b/src/main/daemon/shell-ready.ts @@ -101,11 +101,6 @@ export function resolvePtyShellPath(env: Record): string { return env.SHELL || process.env.SHELL || '/bin/zsh' } -export function shellReadyMarkerComesFromLineEditor(shellPath: string): boolean { - const shellName = pathWin32.basename(basename(shellPath)).toLowerCase() - return shellName === 'bash' || shellName === 'zsh' -} - export function shellPathSupportsPtyStartupBarrier(shellPath: string): boolean { const shellName = pathWin32.basename(basename(shellPath)).toLowerCase() // Why fish: markerless, its startup command is written before fish's reader owns diff --git a/src/main/providers/local-pty-finalize-environment.ts b/src/main/providers/local-pty-finalize-environment.ts index 0f7d3454a09..1b385639e94 100644 --- a/src/main/providers/local-pty-finalize-environment.ts +++ b/src/main/providers/local-pty-finalize-environment.ts @@ -109,6 +109,9 @@ export function finalizeLocalPtySpawnEnvironment(args: { codexStartupCommand !== undefined && supportsPosixShellStartupCommand(shell) ? codexStartupCommand : undefined + // Why no line-editor widening here (unlike the daemon and relay): a Codex + // startup command this provider wraps is run by the wrapper's own prompt + // hook, never written into the PTY, so there is no early write to double-echo. const waitsForShellReady = Boolean(spawn.command) && (!isCodexStartupCommand || codexRequiresShellReady) return getShellLaunchConfig( diff --git a/src/main/shell-prompt-readiness-probe.test.ts b/src/main/shell-prompt-readiness-probe.test.ts index c2c783953ac..56d4b35a06c 100644 --- a/src/main/shell-prompt-readiness-probe.test.ts +++ b/src/main/shell-prompt-readiness-probe.test.ts @@ -3,12 +3,16 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const lineEditorProbe = vi.hoisted(() => vi.fn()) const processReadinessProbe = vi.hoisted(() => vi.fn()) const resolveExecutablePath = vi.hoisted(() => vi.fn((value: string) => Promise.resolve(value))) +const resolveInstalledExecutablePaths = vi.hoisted(() => + vi.fn((): Promise => Promise.resolve([])) +) vi.mock('../shared/pty-slave-line-discipline-echo', () => ({ createPtySlaveLineEditorProbe: () => lineEditorProbe })) vi.mock('../shared/shell-process-readiness', () => ({ readShellProcessReadiness: processReadinessProbe, - resolveShellExecutablePath: resolveExecutablePath + resolveShellExecutablePath: resolveExecutablePath, + resolveInstalledShellExecutablePaths: resolveInstalledExecutablePaths })) import { createShellPromptReadinessProbe } from './shell-prompt-readiness-probe' @@ -19,6 +23,8 @@ describe('shell prompt readiness probe', () => { lineEditorProbe.mockReset() processReadinessProbe.mockReset() resolveExecutablePath.mockClear() + resolveInstalledExecutablePaths.mockClear() + resolveInstalledExecutablePaths.mockResolvedValue([]) }) afterEach(() => { @@ -95,6 +101,99 @@ describe('shell prompt readiness probe', () => { } }) + it('accepts a second installation of the same shell that the pane PATH resolves', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ + executablePath: '/opt/homebrew/bin/bash', + foreground: true + }) + resolveInstalledExecutablePaths.mockResolvedValue(['/bin/bash', '/opt/homebrew/bin/bash']) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + shellCwd: '/work', + shellPathEnv: '/opt/homebrew/bin:/usr/bin:/bin', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(resolveInstalledExecutablePaths).toHaveBeenCalledWith( + 'bash', + '/work', + '/opt/homebrew/bin:/usr/bin:/bin' + ) + expect(onPromptReady).toHaveBeenCalledOnce() + }) + + it('rejects a replacement with the shell basename that the pane PATH cannot reach', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ executablePath: '/tmp/bash', foreground: true }) + resolveInstalledExecutablePaths.mockResolvedValue(['/bin/bash', '/opt/homebrew/bin/bash']) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(onPromptReady).not.toHaveBeenCalled() + }) + + it('does not widen identity when the launched shell path resolves exactly', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ executablePath: '/bin/zsh', foreground: true }) + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/zsh', + getShellPid: () => 42, + onPromptReady: vi.fn(), + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(resolveInstalledExecutablePaths).not.toHaveBeenCalled() + }) + + it('invalidates an alternate-installation result that resolves after disposal', async () => { + const pending: { resolve?: (value: string[]) => void } = {} + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ + executablePath: '/opt/homebrew/bin/bash', + foreground: true + }) + resolveInstalledExecutablePaths.mockImplementation( + () => new Promise((resolve) => (pending.resolve = resolve)) + ) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + probe?.dispose() + pending.resolve?.(['/opt/homebrew/bin/bash']) + await vi.advanceTimersByTimeAsync(0) + + expect(onPromptReady).not.toHaveBeenCalled() + }) + it('does no external work when the ready marker cancels the settle window', async () => { const probe = createShellPromptReadinessProbe({ slavePath: '/dev/ttys048', diff --git a/src/main/shell-prompt-readiness-probe.ts b/src/main/shell-prompt-readiness-probe.ts index 11583fd2de7..3309ac91108 100644 --- a/src/main/shell-prompt-readiness-probe.ts +++ b/src/main/shell-prompt-readiness-probe.ts @@ -1,6 +1,7 @@ import { createPtySlaveLineEditorProbe } from '../shared/pty-slave-line-discipline-echo' import { readShellProcessReadiness, + resolveInstalledShellExecutablePaths, resolveShellExecutablePath } from '../shared/shell-process-readiness' import { @@ -32,6 +33,7 @@ export function createShellPromptReadinessProbe(options: { } const settleMs = options.settleMs ?? SHELL_PROMPT_PROBE_SETTLE_MS const expectedShellName = options.shellPath ? basename(options.shellPath).toLowerCase() : null + const shellCwd = options.shellCwd ?? process.cwd() const outputScanState = createLineEditorReadyOutputScanState() let disposed = false let timer: ReturnType | null = null @@ -52,11 +54,7 @@ export function createShellPromptReadinessProbe(options: { const [shell, expectedPath] = await Promise.all([ readShellProcessReadiness(shellPid), options.shellPath - ? resolveShellExecutablePath( - options.shellPath, - options.shellCwd ?? process.cwd(), - options.shellPathEnv - ) + ? resolveShellExecutablePath(options.shellPath, shellCwd, options.shellPathEnv) : Promise.resolve(null) ]) if (disposed || scheduledGeneration !== generation) { @@ -66,11 +64,28 @@ export function createShellPromptReadinessProbe(options: { !shell?.foreground || !expectedShellName || !expectedPath || - basename(shell.executablePath).toLowerCase() !== expectedShellName || - shell.executablePath !== expectedPath + basename(shell.executablePath).toLowerCase() !== expectedShellName ) { return } + if (shell.executablePath !== expectedPath) { + // Why widen past the launched path: a startup profile that `exec`s a second + // install of the same shell (Homebrew Bash over /bin/bash) keeps the pid but + // loses the wrapper's marker. Only installs this pane's own PATH resolves + // count, so a binary merely *named* bash/zsh outside it stays rejected. + const installedPaths = await resolveInstalledShellExecutablePaths( + expectedShellName, + shellCwd, + options.shellPathEnv + ) + if ( + disposed || + scheduledGeneration !== generation || + !installedPaths.includes(shell.executablePath) + ) { + return + } + } disposed = true options.onPromptReady() } diff --git a/src/relay/pty-handler-startup-command-delivery.test.ts b/src/relay/pty-handler-startup-command-delivery.test.ts index 817ca756385..af2227a9e92 100644 --- a/src/relay/pty-handler-startup-command-delivery.test.ts +++ b/src/relay/pty-handler-startup-command-delivery.test.ts @@ -133,6 +133,73 @@ describe('PtyHandler', () => { } ) + it.skipIf(process.platform === 'win32')( + 'emits shell-ready markers for plain Codex on a line-editor shell', + async () => { + const oldShell = process.env.SHELL + const oldHome = process.env.HOME + const homeDir = mkdtempSync(join(tmpdir(), 'relay-plain-codex-spawn-')) + + process.env.SHELL = '/bin/bash' + process.env.HOME = homeDir + try { + // No prefill flag and no shell-ready hint: the host decides from its own + // shell, because the client cannot see it (#18767). + await dispatcher.callRequest('pty.spawn', { + env: { HOME: homeDir }, + command: 'codex' + }) + } finally { + if (oldShell === undefined) { + delete process.env.SHELL + } else { + process.env.SHELL = oldShell + } + if (oldHome === undefined) { + delete process.env.HOME + } else { + process.env.HOME = oldHome + } + rmSync(homeDir, { recursive: true, force: true }) + } + + const spawnOptions = mockPtySpawn.mock.calls[0]?.[2] as + | { env?: Record } + | undefined + expect(spawnOptions?.env?.ORCA_SHELL_FEATURES).toContain('ready') + vi.advanceTimersByTime(15_000) + expect(handler.retainedStartupCommandCount).toBe(0) + } + ) + + it.skipIf(process.platform === 'win32')( + 'leaves plain Codex unwaited on a shell that emits the marker before its reader', + async () => { + const oldHome = process.env.HOME + const homeDir = mkdtempSync(join(tmpdir(), 'relay-plain-codex-fish-spawn-')) + + process.env.HOME = homeDir + try { + await dispatcher.callRequest('pty.spawn', { + env: { HOME: homeDir, SHELL: '/usr/bin/fish' }, + command: 'codex' + }) + } finally { + if (oldHome === undefined) { + delete process.env.HOME + } else { + process.env.HOME = oldHome + } + rmSync(homeDir, { recursive: true, force: true }) + } + + const spawnOptions = mockPtySpawn.mock.calls[0]?.[2] as + | { env?: Record } + | undefined + expect(spawnOptions?.env?.ORCA_SHELL_FEATURES ?? '').not.toContain('ready') + } + ) + it.skipIf(process.platform === 'win32')( 'emits shell-ready markers for renderer-delivered Codex native prefill commands', async () => { diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 4b7c6dac2d6..d3a79b8a7c8 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -1896,12 +1896,16 @@ export class PtyHandler { isUnattended: launchAgent !== undefined, platform: process.platform }) + // Why the shell is part of the decision here and not on the client: the client + // cannot see which shell this host runs, and plain Codex must still wait where + // the marker rides the line editor rather than double-echoing an early write. const shouldEmitShellReadyMarker = launchCommandHint !== undefined && shouldUseShellReadyStartupDelivery({ command: launchCommandHint, startupCommandDelivery: - params.startupCommandDelivery === 'shell-ready' ? 'shell-ready' : undefined + params.startupCommandDelivery === 'shell-ready' ? 'shell-ready' : undefined, + shellPath: shell }) const managedStartupCommand = shouldProviderDeliverCommand ? command : launchCommandHint // Why: both renderer- and provider-delivered startup commands use this marker; the delivering side strips it from output. diff --git a/src/renderer/src/lib/launch-agent-background-session-remote.test.ts b/src/renderer/src/lib/launch-agent-background-session-remote.test.ts index 6cf8e9b717d..86d3af0f93e 100644 --- a/src/renderer/src/lib/launch-agent-background-session-remote.test.ts +++ b/src/renderer/src/lib/launch-agent-background-session-remote.test.ts @@ -219,6 +219,33 @@ describe('launchAgentBackgroundSession remote runtime and SSH startup delivery', } }) + // #18767: a plain Codex launch carries no shell-ready hint, but the remote host + // still arms the marker for it, so writing early would display the launch twice. + it('waits for shell-ready for a promptless SSH background Codex launch', async () => { + vi.useFakeTimers() + try { + state.repos = [{ id: 'repo-1', connectionId: 'ssh-1', path: '/repo' }] + const { launchAgentBackgroundSession } = await import('./launch-agent-background-session') + + await launchAgentBackgroundSession({ agent: 'codex', worktreeId: 'wt-1' }) + const dataSidecar = mockSubscribeToPtyData.mock.calls[0]?.[1] as (data: string) => void + dataSidecar('user@remote repo % ') + vi.advanceTimersByTime(1_400) + + expect(mockWrite).not.toHaveBeenCalled() + + dataSidecar('\x1b]777;orca-shell-ready\x07user@remote repo % ') + vi.advanceTimersByTime(50) + + expect(mockWrite).toHaveBeenCalledWith( + 'pty-1', + "codex '--dangerously-bypass-approvals-and-sandbox'\r" + ) + } finally { + vi.useRealTimers() + } + }) + it('falls back when an SSH shell produces no observable startup data', async () => { vi.useFakeTimers() try { diff --git a/src/renderer/src/lib/launch-agent-background-session.ts b/src/renderer/src/lib/launch-agent-background-session.ts index 6bc6c65e420..21eac21ce9e 100644 --- a/src/renderer/src/lib/launch-agent-background-session.ts +++ b/src/renderer/src/lib/launch-agent-background-session.ts @@ -28,8 +28,10 @@ import { subscribeToRuntimeTerminalData, toRemoteRuntimePtyId } from '@/runtime/runtime-terminal-stream' -import { createSshBackgroundStartupDelivery } from '@/lib/ssh-background-startup-delivery' -import { shouldUseShellReadyStartupDelivery } from '../../../shared/codex-startup-delivery' +import { + createSshBackgroundStartupDelivery, + sshBackgroundLaunchWaitsForShellReady +} from '@/lib/ssh-background-startup-delivery' import { isMainTerminalSideEffectAuthorityForPty } from '@/components/terminal-pane/terminal-side-effect-facts-handler' import { resolveLocalWindowsAgentStartupShell } from '../../../shared/windows-terminal-shell' import { runBestEffortAgentBackgroundCleanups } from '@/lib/agent-background-session-cleanup' @@ -115,11 +117,7 @@ export async function launchAgentBackgroundSession( const sshStartupDelivery = createSshBackgroundStartupDelivery({ command: sshConnectionId ? startupPlan.launchCommand : null, waitForShellReady: - Boolean(sshConnectionId) && - shouldUseShellReadyStartupDelivery({ - command: startupPlan.launchCommand, - startupCommandDelivery: startupPlan.startupCommandDelivery - }), + Boolean(sshConnectionId) && sshBackgroundLaunchWaitsForShellReady(startupPlan), write: (ptyId, data) => window.api.pty.write(ptyId, data) }) // Route by the worktree's owner host, not the focused runtime. diff --git a/src/renderer/src/lib/ssh-background-startup-delivery.test.ts b/src/renderer/src/lib/ssh-background-startup-delivery.test.ts index ee5248cf2fd..1c9a8e6486f 100644 --- a/src/renderer/src/lib/ssh-background-startup-delivery.test.ts +++ b/src/renderer/src/lib/ssh-background-startup-delivery.test.ts @@ -18,6 +18,25 @@ function createDelivery(): { } } +// Bracketed paste only wraps multiline submissions, so the marker's effect on it +// is only observable through a command that carries a newline. +const MULTILINE_COMMAND = 'codex "run the\nautomation"' + +function createMultilineDelivery(waitForShellReady: boolean): { + delivery: ReturnType + write: ReturnType +} { + const write = vi.fn() + return { + delivery: createSshBackgroundStartupDelivery({ + command: MULTILINE_COMMAND, + waitForShellReady, + write + }), + write + } +} + beforeEach(() => { vi.useFakeTimers() }) @@ -89,4 +108,28 @@ describe('createSshBackgroundStartupDelivery shell-ready fallback', () => { expect(write).toHaveBeenCalledTimes(1) }) + + // #18767: the marker is what proves the host wrapped the shell and armed + // bracketed paste. A fallback release means it never did. + it('uses bracketed paste only after the marker actually arrived', () => { + const { delivery, write } = createMultilineDelivery(true) + + delivery.armFallback('pty-1') + delivery.handleData(`${SHELL_READY}user@remote repo % `) + vi.advanceTimersByTime(50) + + expect(write.mock.calls[0]?.[1]).toContain('\x1b[200~') + }) + + it('submits raw when the wait ends at the fallback instead of the marker', () => { + const { delivery, write } = createMultilineDelivery(true) + + delivery.armFallback('pty-1') + delivery.handleData('user@remote repo % ') + vi.advanceTimersByTime(1_550) + vi.advanceTimersByTime(50) + + expect(write).toHaveBeenCalledTimes(1) + expect(write.mock.calls[0]?.[1]).not.toContain('\x1b[200~') + }) }) diff --git a/src/renderer/src/lib/ssh-background-startup-delivery.ts b/src/renderer/src/lib/ssh-background-startup-delivery.ts index 1c11254d6bb..108e368a27d 100644 --- a/src/renderer/src/lib/ssh-background-startup-delivery.ts +++ b/src/renderer/src/lib/ssh-background-startup-delivery.ts @@ -2,8 +2,34 @@ import { createShellReadyMarkerScanState, scanForShellReadyMarker } from '@/components/terminal-pane/shell-ready-marker-scan' +import { + isCodexStartupCommand, + shouldUseShellReadyStartupDelivery, + type StartupCommandDelivery +} from '../../../shared/codex-startup-delivery' import { buildStartupCommandSubmission } from '../../../shared/startup-command-submission' +/** + * Why every Codex launch waits and not only the prompt-carrying ones: the remote + * shell is the host's to know, and it arms the ready marker for plain Codex too + * (#18767). On such a host the wait ends at the prompt and costs nothing. On one + * that never publishes a marker -- fish, sh, Windows, or a host predating #18767 -- + * the fallback below releases instead, at the same price prompt-carrying Codex + * already paid there. + */ +export function sshBackgroundLaunchWaitsForShellReady(startupPlan: { + launchCommand: string | null | undefined + startupCommandDelivery?: StartupCommandDelivery +}): boolean { + return ( + isCodexStartupCommand(startupPlan.launchCommand) || + shouldUseShellReadyStartupDelivery({ + command: startupPlan.launchCommand, + startupCommandDelivery: startupPlan.startupCommandDelivery + }) + ) +} + const SSH_SHELL_READY_STARTUP_FALLBACK_MS = 1500 // Why: a remote shell that has not emitted a single byte is still booting — // /etc/profile plus nvm/conda/pyenv over a cold link routinely needs more than @@ -31,6 +57,10 @@ export function createSshBackgroundStartupDelivery( let pendingCommand = options.command let lastPtyId: string | null = null let startupShellReady = !options.waitForShellReady + // Why tracked apart from `startupShellReady`: only an observed marker proves the + // host wrapped the shell and armed bracketed paste. A fallback release means the + // host shell never published one, so the raw submit is the only safe form. + let markerObserved = false const markerScan = options.waitForShellReady ? createShellReadyMarkerScanState() : null let injectTimer: ReturnType | null = null let fallbackTimer: ReturnType | null = null @@ -53,6 +83,7 @@ export function createSshBackgroundStartupDelivery( return } startupShellReady = true + markerObserved = true clearFallbackTimer() if (pendingCommand && lastPtyId) { schedule(lastPtyId) @@ -100,14 +131,14 @@ export function createSshBackgroundStartupDelivery( // Why: the SSH relay treats spawn.command as metadata for interactive // PTYs; hidden automation tabs still submit the command themselves. // Why bracketed paste: multiline prompts are pasted literally only when we - // synchronized on the Orca shell-ready marker (waitForShellReady) — that - // is the bash/zsh overlay with bracketed-paste mode armed. Submit with CR - // since the relay drives a remote shell. + // synchronized on the Orca shell-ready marker — that is the bash/zsh overlay + // with bracketed-paste mode armed. Submit with CR since the relay drives a + // remote shell. options.write( ptyId, buildStartupCommandSubmission(command, { submit: '\r', - bracketedPasteSafe: options.waitForShellReady + bracketedPasteSafe: markerObserved }) ) }, 50) diff --git a/src/shared/codex-startup-delivery.test.ts b/src/shared/codex-startup-delivery.test.ts index e7788f778fe..87e65f7514c 100644 --- a/src/shared/codex-startup-delivery.test.ts +++ b/src/shared/codex-startup-delivery.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import { hasCodexNativeDraftFlag } from './codex-startup-delivery' +import { + hasCodexNativeDraftFlag, + shouldUseShellReadyStartupDelivery +} from './codex-startup-delivery' describe('hasCodexNativeDraftFlag', () => { it('matches Codex --prefill option tokens', () => { @@ -26,3 +29,42 @@ describe('hasCodexNativeDraftFlag', () => { expect(hasCodexNativeDraftFlag('codex --model gpt-5')).toBe(false) }) }) + +describe('shouldUseShellReadyStartupDelivery', () => { + it('honours an explicit shell-ready hint whatever the command', () => { + expect( + shouldUseShellReadyStartupDelivery({ + command: 'claude', + startupCommandDelivery: 'shell-ready' + }) + ).toBe(true) + }) + + it('keeps plain Codex on the fast path when the shell is unknown', () => { + expect(shouldUseShellReadyStartupDelivery({ command: 'codex' })).toBe(false) + }) + + it('waits for plain Codex on shells that publish the marker from the line editor', () => { + expect(shouldUseShellReadyStartupDelivery({ command: 'codex', shellPath: '/bin/bash' })).toBe( + true + ) + expect( + shouldUseShellReadyStartupDelivery({ command: 'codex', shellPath: '/opt/homebrew/bin/zsh' }) + ).toBe(true) + }) + + it('leaves plain Codex unwaited on shells that emit the marker before the reader', () => { + expect( + shouldUseShellReadyStartupDelivery({ command: 'codex', shellPath: '/usr/bin/fish' }) + ).toBe(false) + }) + + it('does not change non-Codex commands, which their transports already wait for', () => { + expect(shouldUseShellReadyStartupDelivery({ command: 'claude', shellPath: '/bin/bash' })).toBe( + false + ) + expect(shouldUseShellReadyStartupDelivery({ command: undefined, shellPath: '/bin/bash' })).toBe( + false + ) + }) +}) diff --git a/src/shared/codex-startup-delivery.ts b/src/shared/codex-startup-delivery.ts index a45337dd0be..538b3befa19 100644 --- a/src/shared/codex-startup-delivery.ts +++ b/src/shared/codex-startup-delivery.ts @@ -1,4 +1,5 @@ import { recognizeAgentProcessFromCommandLine } from './agent-process-recognition' +import { shellReadyMarkerComesFromLineEditor } from './shell-ready-marker-timing' export type StartupCommandDelivery = 'fast' | 'shell-ready' @@ -74,9 +75,23 @@ export function hasCodexNativeDraftFlag(command: string | null | undefined): boo ) } +export function isCodexStartupCommand(command: string | null | undefined): boolean { + return recognizeAgentProcessFromCommandLine(command)?.agent === 'codex' +} + export function shouldUseShellReadyStartupDelivery(args: { command: string | null | undefined startupCommandDelivery?: StartupCommandDelivery + /** The shell that will run the command, when the deciding side knows it. Plain Codex + * waits for the handshake on shells that publish the marker from their line editor: + * there the wait ends at the prompt, while an early write double-echoes the launch. */ + shellPath?: string }): boolean { - return args.startupCommandDelivery === 'shell-ready' || hasCodexNativeDraftFlag(args.command) + return ( + args.startupCommandDelivery === 'shell-ready' || + hasCodexNativeDraftFlag(args.command) || + (args.shellPath !== undefined && + shellReadyMarkerComesFromLineEditor(args.shellPath) && + isCodexStartupCommand(args.command)) + ) } diff --git a/src/shared/shell-process-readiness.test.ts b/src/shared/shell-process-readiness.test.ts index 9c3abad9783..a31f8ba7737 100644 --- a/src/shared/shell-process-readiness.test.ts +++ b/src/shared/shell-process-readiness.test.ts @@ -2,7 +2,11 @@ import { mkdir, mkdtemp, rm, symlink } from 'node:fs/promises' import { tmpdir } from 'node:os' import { basename, dirname, join } from 'node:path' import { describe, expect, it } from 'vitest' -import { parseDarwinExecutablePath, resolveShellExecutablePath } from './shell-process-readiness' +import { + parseDarwinExecutablePath, + resolveInstalledShellExecutablePaths, + resolveShellExecutablePath +} from './shell-process-readiness' describe('shell process readiness', () => { it('extracts the primary text image from macOS lsof output', () => { @@ -70,4 +74,46 @@ describe('shell process readiness', () => { } } ) + + it.skipIf(process.platform === 'win32')( + 'lists every PATH installation of a shell name, deduplicated and canonical', + async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-installed-shells-')) + const first = join(root, 'first') + const second = join(root, 'second') + const missing = join(root, 'missing') + await mkdir(first) + await mkdir(second) + await symlink(process.execPath, join(first, 'shell-name')) + await symlink(process.execPath, join(second, 'shell-name')) + try { + const canonical = await resolveShellExecutablePath(process.execPath, root, '') + await expect( + resolveInstalledShellExecutablePaths('shell-name', root, `${first}:${missing}:${second}`) + ).resolves.toEqual([canonical]) + } finally { + await rm(root, { recursive: true, force: true }) + } + } + ) + + it.skipIf(process.platform === 'win32')( + 'omits a same-name executable that no PATH entry reaches', + async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-offpath-shell-')) + const onPath = join(root, 'bin') + const offPath = join(root, 'dropped') + await mkdir(onPath) + await mkdir(offPath) + await symlink(process.execPath, join(onPath, 'shell-name')) + await symlink(process.execPath, join(offPath, 'shell-name')) + try { + await expect( + resolveInstalledShellExecutablePaths('shell-name', root, onPath) + ).resolves.not.toContain(join(offPath, 'shell-name')) + } finally { + await rm(root, { recursive: true, force: true }) + } + } + ) }) diff --git a/src/shared/shell-process-readiness.ts b/src/shared/shell-process-readiness.ts index ebb94c13eb5..c9976b0d03e 100644 --- a/src/shared/shell-process-readiness.ts +++ b/src/shared/shell-process-readiness.ts @@ -56,12 +56,12 @@ export async function readShellProcessReadiness( : null } -export async function resolveShellExecutablePath( +function shellExecutableCandidates( shellPath: string, cwd: string, pathEnv: string | undefined -): Promise { - const candidates = shellPath.includes('/') +): string[] { + return shellPath.includes('/') ? [isAbsolute(shellPath) ? shellPath : resolve(cwd, shellPath)] : ( pathEnv ?? @@ -69,14 +69,42 @@ export async function resolveShellExecutablePath( ) .split(delimiter) .map((entry) => resolve(isAbsolute(entry) ? entry : resolve(cwd, entry), shellPath)) - for (const candidate of candidates) { - try { - await access(candidate, constants.X_OK) - const canonicalPath = await realpath(candidate) - if ((await stat(canonicalPath)).isFile()) { - return canonicalPath - } - } catch {} +} + +async function canonicalizeExecutable(candidate: string): Promise { + try { + await access(candidate, constants.X_OK) + const canonicalPath = await realpath(candidate) + return (await stat(canonicalPath)).isFile() ? canonicalPath : null + } catch { + return null + } +} + +export async function resolveShellExecutablePath( + shellPath: string, + cwd: string, + pathEnv: string | undefined +): Promise { + for (const candidate of shellExecutableCandidates(shellPath, cwd, pathEnv)) { + const canonicalPath = await canonicalizeExecutable(candidate) + if (canonicalPath) { + return canonicalPath + } } return null } + +/** Every canonical executable `shellName` names on `pathEnv` — the installations a + * startup profile could legitimately `exec` into, and nothing a dropped-in binary + * outside the search path can reach. `shellName` must be a bare name. */ +export async function resolveInstalledShellExecutablePaths( + shellName: string, + cwd: string, + pathEnv: string | undefined +): Promise { + const canonicalPaths = await Promise.all( + shellExecutableCandidates(shellName, cwd, pathEnv).map(canonicalizeExecutable) + ) + return [...new Set(canonicalPaths.filter((path): path is string => path !== null))] +} diff --git a/src/shared/shell-ready-marker-timing.ts b/src/shared/shell-ready-marker-timing.ts new file mode 100644 index 00000000000..7c73afb4c31 --- /dev/null +++ b/src/shared/shell-ready-marker-timing.ts @@ -0,0 +1,18 @@ +/** + * When in a shell's startup Orca's OSC 777 ready marker is published. + * + * Why this is a decision of its own: it is what separates "waiting for the marker + * is free" from "waiting for the marker costs the user real startup latency", and + * all three transports (daemon, relay, local provider) have to answer it the same + * way or a startup command is delivered twice on one of them. + */ + +/** + * True when the marker rides the shell's line editor (zsh `precmd`, bash + * `PROMPT_COMMAND`), so it arrives at the same moment the prompt can accept input. + * Every other wrapped shell emits it from startup, ahead of the reader. + */ +export function shellReadyMarkerComesFromLineEditor(shellPath: string): boolean { + const shellName = shellPath.replace(/\\/g, '/').split('/').pop()?.toLowerCase() ?? '' + return shellName === 'bash' || shellName === 'zsh' +} From b0c67eaf883d2ed31e7496a98124b75971588a2a Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:31:39 -0700 Subject: [PATCH 10/26] feat(mobile): port the restructured native-chat turn status and live tool progress (#18761) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(mobile): port the restructured native-chat turn status and live tool progress Mobile chat had a single static "Agent is working" row and no live tool activity, while the desktop restructure (#17597, #18705) replaced that with a per-turn status row and a running-tool label. This brings mobile to parity and puts the derivation in one place instead of two. Shared (new, pure, RN-safe — desktop uses them as i18n fallbacks, mobile directly, matching the native-chat-empty-state pattern): - `native-chat-turn-status.ts`: duration formatting, label selection, the turn-timing state machine, and the active/settled split. - `native-chat-tool-activity.ts`: command-tool classification, the running-tool label descriptor, and running-call selection. Desktop now consumes both; `NativeChatWorkingStatus`, `NativeChatToolRun` and `use-native-chat-turn-status` keep their existing behavior and strings. Mobile gains the "Thinking" / "Working for 12s" / "Worked for 3m 4s" row with a caret that discloses the turn's tool activity, the pulsing "Running npm test" row with terminal-vs-wrench glyphs, and desktop's rule that a completed turn's tool run hides behind the turn caret. The bridge lane is untouched and keeps its three-dot indicator. Headings, quotes, code, lists and table cells are now selectable. Files at their max-lines cap were split rather than bumped: the tool-run subtree, the prompt card, the session-lane wiring, and the turn-disclosure state each move to their own module. * perf(mobile): stop the turn-status rows from re-rendering the whole transcript A streaming turn re-renders the chat list many times a second. The disclosure wiring handed every row a fresh status object and a fresh toggle closure on each of those renders, so `MobileNativeChatMessage`'s memo never held and every visible row re-rendered per tick — including settled turns that had not changed. Memoize the status selection on the timing map, and keep one stable toggle handler per turn (pruned when a turn leaves the transcript) attached only to the settled rows that can actually disclose anything. Now only the live turn's row changes identity while the agent works. * fix(mobile): keep the turn clock running when the optimistic echo is replaced An accepted send renders as `pending-N` until the transcript echo lands under its real message id. That flips the active turn key mid-turn, and the timing reducer treated the new key as a new turn — so a turn that had reached "Working for 8s" visibly restarted at "Working for 0s". The reducer now carries the start over when the previous key names a turn that has since left the transcript, which is exactly the echo-replacement case. A genuinely new turn (the previous key still in the transcript) and a turn that had already settled both keep their own clock; both are pinned by tests. Desktop does not pass the new key and is unaffected. * fix(mobile): keep the Tools toggle working on settled turns Hiding a settled turn's tool run behind the turn caret (desktop parity) also made the composer's global Tools control a no-op on every completed turn: the run it wanted to expand was not rendered at all. Let that toggle override the hiding, so it still reveals every run at once the way it did before. * fix(mobile): re-key the turn timing instead of only carrying its start The previous fix carried the start forward only while the turn was still working. When the transcript echo landed after the turn had already settled, the new key inherited nothing, the settled timing was pruned with the old key, and the turn's "Worked for N" row disappeared entirely. Move the timing onto the new key instead, which covers both orderings: an in-flight turn keeps counting from its original start (and later settles against it), and an already-settled turn keeps its duration. Both orderings are pinned. * test(mobile): pin the structured turn-status wiring at the view level Emulator QA could not reach the structured lane (mobile's Create Tab -> Codex falls back to a terminal tab when agentSession.createSupport says unsupported), so the view's own lane wiring had no coverage — the one seam between the shared turn-timing reducer and the rendered rows. Assert what the view hands each row: the live user turn gets a status object and the three-dot indicator is gone on the structured lane; the bridge lane keeps the indicator and gets no status; a finished turn settles to a numeric duration with a toggle; and an assistant row never carries a status row of its own. * fix(mobile): isolate structured chat turn state * fix(mobile): let the capability RPC actually store what a phone advertises `runtime.clientCapabilities.update` records the advertised set by assigning `authenticatedSocket.clientCapabilities`, but the socket handed to the dispatcher defined that property with a getter only. In strict mode the assignment throws `TypeError: Cannot set property clientCapabilities ... which has only a getter`, so the RPC answered `runtime_error` and the set was never stored. The consequence is not subtle: `supportsStructuredAgentSessions` requires the capability, so `projectSessionTabAgentStatus` removed every `agent-session` tab from a phone that had advertised it correctly. A paired phone saw ZERO tabs on a worktree whose only tab was a structured Codex chat — structured native chat was unreachable on mobile over this transport, not just missing its new turn UI. Give the socket a setter that writes through to the channel, which already owns the set for the connection's lifetime, so later requests on the same socket see it. Found while trying to capture emulator screenshots of the turn-status port: two full QA runs reported the new UI "missing" because the phone could only ever get a bridge/PTY tab. * fix(mobile): carry the turn key instead of caching a handler in a ref Builds on the scope-isolation fix: that kept (and extended) a ref that is written during render — once to memoize a per-turn handler, once to prune dead turns, once to reset on a scope change. React Doctor's "Ref mutated during render" is what CI's `check:react-doctor:changed` was failing on (x2), and on mobile it is a real hazard rather than a style note: react-freeze discards renders, and a discarded render would leave the cache mutated. Pass the settled turn's key down the row instead and let it call one stable handler with it. That preserves both properties the cache was bought for — per scope isolation, and identity stability so a streaming transcript does not defeat the row's memo — with no ref writes and no pruning to get wrong. The scope-keyed expanded set and the 128-turn cap are untouched; their tests move to the new contract and one now pins handler identity across a re-render. Note for future changes here: `check:code-quality:changed` does NOT cover this. CI additionally runs the standalone react-doctor CLI, which has rules the oxlint plugin config does not enable. * fix: ship native chat status translations * test(native-chat): pin the shared copy against the English catalog The shared constants are desktop's i18n fallback and mobile's actually-rendered string. If one changes without the other, desktop keeps rendering en.json while mobile renders the constant — and nothing fails, because a fallback is only used when the key is missing. That silent divergence is the exact thing the shared module exists to prevent, and it is now reachable precisely because these strings are runtime-required rather than statically extracted. Assert every key in both shared copy objects matches en.json byte for byte, plus the interpolation placeholders the catalog interpolates on. --------- Co-authored-by: Merge Sim --- mobile/src/components/MobileMarkdown.tsx | 15 +- .../session/MobileNativeChatMessage.test.ts | 124 ++++++- .../src/session/MobileNativeChatMessage.tsx | 307 +++++----------- .../src/session/MobileNativeChatOverlay.tsx | 1 + .../session/MobileNativeChatPromptCard.tsx | 74 ++++ .../src/session/MobileNativeChatToolRun.tsx | 250 +++++++++++++ .../MobileNativeChatTurnStatus.test.ts | 112 ++++++ .../session/MobileNativeChatTurnStatus.tsx | 117 ++++++ .../src/session/MobileNativeChatView.test.ts | 110 ++++++ mobile/src/session/MobileNativeChatView.tsx | 86 ++--- .../mobile-native-chat-controller-contract.ts | 2 + .../mobile-native-chat-message-styles.ts | 12 + .../use-mobile-native-chat-controller.ts | 39 +- .../use-mobile-native-chat-session-lane.ts | 59 +++ ...obile-native-chat-turn-disclosure.test.tsx | 153 ++++++++ .../use-mobile-native-chat-turn-disclosure.ts | 126 +++++++ .../use-mobile-native-chat-turn-status.ts | 105 ++++++ .../runtime/rpc/mobile-socket-wiring.test.ts | 52 +++ src/main/runtime/rpc/mobile-socket-wiring.ts | 7 + .../native-chat/NativeChatToolRun.tsx | 73 ++-- .../native-chat/NativeChatWorkingStatus.tsx | 65 ++-- ...e-chat-shared-copy-matches-catalog.test.ts | 43 +++ .../use-native-chat-turn-status.ts | 115 ++---- .../src/i18n/en-runtime-required.json | 13 +- src/shared/native-chat-tool-activity.test.ts | 109 ++++++ src/shared/native-chat-tool-activity.ts | 97 +++++ src/shared/native-chat-turn-status.test.ts | 336 ++++++++++++++++++ src/shared/native-chat-turn-status.ts | 210 +++++++++++ 28 files changed, 2344 insertions(+), 468 deletions(-) create mode 100644 mobile/src/session/MobileNativeChatPromptCard.tsx create mode 100644 mobile/src/session/MobileNativeChatToolRun.tsx create mode 100644 mobile/src/session/MobileNativeChatTurnStatus.test.ts create mode 100644 mobile/src/session/MobileNativeChatTurnStatus.tsx create mode 100644 mobile/src/session/use-mobile-native-chat-session-lane.ts create mode 100644 mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx create mode 100644 mobile/src/session/use-mobile-native-chat-turn-disclosure.ts create mode 100644 mobile/src/session/use-mobile-native-chat-turn-status.ts create mode 100644 src/renderer/src/components/native-chat/native-chat-shared-copy-matches-catalog.test.ts create mode 100644 src/shared/native-chat-tool-activity.test.ts create mode 100644 src/shared/native-chat-tool-activity.ts create mode 100644 src/shared/native-chat-turn-status.test.ts create mode 100644 src/shared/native-chat-turn-status.ts diff --git a/mobile/src/components/MobileMarkdown.tsx b/mobile/src/components/MobileMarkdown.tsx index cc88c01e564..2f5b52cfe18 100644 --- a/mobile/src/components/MobileMarkdown.tsx +++ b/mobile/src/components/MobileMarkdown.tsx @@ -192,6 +192,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return ( {renderInline(block.text, onOpenFile)} @@ -201,7 +202,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile if (block.type === 'quote') { return ( - {renderInline(block.text, onOpenFile)} + + {renderInline(block.text, onOpenFile)} + ) } @@ -223,7 +226,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return ( {block.language ? {block.language} : null} - {block.text} + + {block.text} + ) } @@ -251,7 +256,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleHeaders.map((header, cellIndex) => ( - + {renderInline(header, onOpenFile)} ))} @@ -259,7 +264,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleRows.map((row, rowIndex) => ( {visibleHeaders.map((_, cellIndex) => ( - + {renderInline(row[cellIndex] ?? '', onOpenFile)} ))} @@ -290,7 +295,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile ? '[x]' : '[ ]'} - + {renderInline(item.text, onOpenFile)} diff --git a/mobile/src/session/MobileNativeChatMessage.test.ts b/mobile/src/session/MobileNativeChatMessage.test.ts index 10b96cc3d20..677b76d240c 100644 --- a/mobile/src/session/MobileNativeChatMessage.test.ts +++ b/mobile/src/session/MobileNativeChatMessage.test.ts @@ -6,11 +6,21 @@ import type { NativeChatMessage } from '../../../src/shared/native-chat-types' vi.mock('react-native', async () => { const React = await import('react') + const Text = ({ children, ...props }: { children?: unknown }): unknown => + React.createElement('Text', props, children) return { + Animated: { + Text, + Value: class { + setValue(): void {} + }, + loop: (animation: unknown) => animation, + sequence: () => ({ start: vi.fn(), stop: vi.fn() }), + timing: () => ({ start: vi.fn(), stop: vi.fn() }) + }, Image: 'Image', Pressable: 'Pressable', - Text: ({ children, ...props }: { children?: unknown }) => - React.createElement('Text', props, children), + Text, View: ({ children, ...props }: { children?: unknown }) => React.createElement('View', props, children), StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } @@ -21,7 +31,10 @@ vi.mock('lucide-react-native', () => ({ ArrowUp: 'ArrowUp', ChevronDown: 'ChevronDown', Copy: 'Copy', - SquareChevronRight: 'SquareChevronRight' + SquareChevronRight: 'SquareChevronRight', + SquareTerminal: 'SquareTerminal', + Wrench: 'Wrench', + ChevronRight: 'ChevronRight' })) vi.mock('../components/MobileMarkdown', () => ({ MobileMarkdown: 'MobileMarkdown' })) @@ -45,7 +58,18 @@ describe('MobileNativeChatMessage', () => { function render( message: NativeChatMessage, - props: { toolsExpanded?: boolean } = {} + props: { + toolsExpanded?: boolean + structuredActivityUi?: boolean + activeTurnIsWorking?: boolean + turnExpanded?: boolean + turnStatus?: { + startedAt: number | null + thinking: boolean + workedSeconds: number | null + } | null + onToggleTurn?: () => void + } = {} ): ReactTestRenderer { act(() => { renderer = create(createElement(MobileNativeChatMessage, { message, ...props })) @@ -152,4 +176,96 @@ describe('MobileNativeChatMessage', () => { expect(tree.root.findAllByType('ChevronDown' as never)).toHaveLength(1) expect(tree.root.findAllByType('SquareChevronRight' as never)).toHaveLength(1) }) + + describe('structured activity UI', () => { + const runningCall = { + type: 'tool-call' as const, + name: 'Bash', + input: { command: 'npm test' }, + state: 'running' as const + } + const settledCall = { + type: 'tool-call' as const, + name: 'Read', + input: { file_path: 'a/b.ts' }, + state: 'completed' as const + } + + it('shows the live tool label with a terminal glyph while a command runs', () => { + const tree = render(toolMessage([runningCall]), { + structuredActivityUi: true, + activeTurnIsWorking: true + }) + expect(textIn(tree.root)).toContain('Running npm test') + expect(tree.root.findAllByType('SquareTerminal' as never)).toHaveLength(1) + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) + }) + + it('uses the wrench glyph for a non-command tool', () => { + const tree = render( + toolMessage([ + { type: 'tool-call', name: 'Read', input: { file_path: 'a/b.ts' }, state: 'running' } + ]), + { structuredActivityUi: true, activeTurnIsWorking: true } + ) + expect(textIn(tree.root)).toContain('Running Read a/b.ts') + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(1) + }) + + it('falls back to the collapsed count row once the run settles', () => { + const tree = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: true + }) + expect(textIn(tree.root)).not.toContain('Running Read a/b.ts') + expect(textIn(tree.root)).toContain('1×') + }) + + it("hides a completed turn's activity until the turn caret discloses it", () => { + const collapsed = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false + }) + expect(textIn(collapsed.root)).not.toContain('1×') + act(() => collapsed.unmount()) + + const disclosed = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false, + turnExpanded: true + }) + expect(textIn(disclosed.root)).toContain('1×') + }) + + it('lets the global Tools toggle reveal a hidden settled run', () => { + // Otherwise the composer's Tools control is a no-op on every settled turn. + const tree = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false, + toolsExpanded: true + }) + expect(textIn(tree.root)).toContain('1\u00d7') + }) + + it('keeps the bridge lane on its always-visible tool run', () => { + const tree = render(toolMessage([settledCall]), { activeTurnIsWorking: false }) + expect(textIn(tree.root)).toContain('1×') + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) + }) + + it('renders the turn status row under a user message', () => { + const tree = render(userMessage([{ type: 'text', text: 'go' }]), { + structuredActivityUi: true, + turnStatus: { startedAt: Date.now(), thinking: true, workedSeconds: null } + }) + expect(textIn(tree.root)).toContain('Thinking') + }) + + it('does not render a turn status row without one', () => { + const tree = render(userMessage([{ type: 'text', text: 'go' }]), { + structuredActivityUi: true + }) + expect(textIn(tree.root)).toEqual(['go']) + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatMessage.tsx b/mobile/src/session/MobileNativeChatMessage.tsx index 0a676061e5b..cc6c086b1f3 100644 --- a/mobile/src/session/MobileNativeChatMessage.tsx +++ b/mobile/src/session/MobileNativeChatMessage.tsx @@ -1,139 +1,20 @@ import { memo, useEffect, useRef, useState } from 'react' import { Image, Pressable, Text, View } from 'react-native' import * as Clipboard from 'expo-clipboard' -import { ArrowUp, ChevronDown, Copy, SquareChevronRight } from 'lucide-react-native' -import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff' -import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff' -import { pairToolBlocks, splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' -import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold' -import { - createToolInputDisplay, - summarizeToolRun, - truncateToolDetail -} from '../../../src/shared/native-chat-tool-summary' +import { ArrowUp, Copy } from 'lucide-react-native' +import { splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' +import { selectActiveToolCall } from '../../../src/shared/native-chat-tool-activity' import { isImageRefBlock, isTextBlock } from '../../../src/shared/native-chat-types' import type { NativeChatBlock, NativeChatMessage } from '../../../src/shared/native-chat-types' import { MobileMarkdown } from '../components/MobileMarkdown' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' +import { ToolRun } from './MobileNativeChatToolRun' +import type { NativeChatTurnStatus } from './use-mobile-native-chat-turn-status' import { colors } from '../theme/mobile-theme' import { isRenderableImageUri } from './mobile-native-chat-image-preview' import { styles, TEXT_SIZE } from './mobile-native-chat-message-styles' import { nativeChatMessageText } from './mobile-native-chat-message-text' -const MAX_VISIBLE_TOOL_PAIRS = 6 -const MAX_TOOL_RUN_DIFF_ROWS = 240 - -function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element { - return ( - - {lines.map((line, i) => ( - - {line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '} - {line.text} - - ))} - - ) -} - -/** A single inline tool line — `▸ ToolName preview` — that expands in place to - * show the call's diff/input or the result's body. Mirrors the reference design - * where tool calls read as flat lines in the conversation, not boxed blocks. */ -function ResultBody({ - output, - isError, - diff -}: { - output: string - isError?: boolean - diff: DiffLine[] | null -}): React.JSX.Element { - if (diff) { - return - } - return ( - - {truncateToolDetail(output)} - - ) -} - -/** One request: a tool call and its result rendered together as a single - * expandable line. `defaultExpanded` lets the group toggle open every line. */ -function ToolLine({ - pair, - defaultExpanded, - diffLineLimit, - onOpenFile -}: { - pair: ToolPair - defaultExpanded: boolean - diffLineLimit: number - onOpenFile?: (relativePath: string) => void -}): React.JSX.Element { - const [expanded, setExpanded] = useState(defaultExpanded) - const { call, result } = pair - const name = call ? call.name : 'Result' - const inputDisplay = call ? createToolInputDisplay(call.input) : null - const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? '' - // Why: collapsed tool rows are the common path; defer bounded diff parsing - // and detail formatting until the user asks to reveal the detail. - const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null - const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null - const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined - const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true - // The group toggle opens every line at once, bypassing the tap guard, so the - // panel has to consult it too — else a detail-less row echoes its own label - // under itself and no tap can dismiss it. - const showDetail = hasDetail && expanded - // A tool that targets a file (Read/Edit/Write…) renders its preview as a - // tappable link that opens the file, independent of the line's expand tap. - const filePath = inputDisplay?.filePath ?? null - const openable = filePath !== null && onOpenFile !== undefined - return ( - - hasDetail && setExpanded((v) => !v)} - hitSlop={6} - > - {showDetail ? ( - - ) : ( - - )} - {name} - {preview ? ( - onOpenFile!(filePath!) : undefined} - suppressHighlighting={!openable} - > - {preview} - - ) : null} - - {showDetail ? ( - - {callDiff ? : null} - {callDetail ? {callDetail} : null} - {result ? ( - - ) : null} - - ) : null} - - ) -} - function Prose({ block, invert, @@ -150,7 +31,9 @@ function Prose({ // markdown renderer's light-on-dark palette. if (invert) { return ( - {block.text} + + {block.text} + ) } return ( @@ -180,67 +63,6 @@ function Prose({ return null } -/** A run of a message's tool calls/results, collapsed to a one-line summary that - * expands to the individual inline tool lines. `defaultExpanded` lets the global - * toolbar toggle drive every run at once while still allowing per-run override. */ -function ToolRun({ - blocks, - defaultExpanded, - trailing, - onOpenFile -}: { - blocks: NativeChatBlock[] - defaultExpanded: boolean - trailing?: React.ReactNode - onOpenFile?: (relativePath: string) => void -}): React.JSX.Element { - const [open, setOpen] = useState(defaultExpanded) - const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS) - const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1))) - let callCount = 0 - for (const block of blocks) { - if (block.type === 'tool-call') { - callCount++ - } - } - callCount ||= pairs.length - const summary = summarizeToolRun(blocks) - return ( - - - setOpen((v) => !v)} hitSlop={6}> - {open ? ( - - ) : ( - - )} - {callCount}× - - {summary || `${callCount} tool ${callCount === 1 ? 'call' : 'calls'}`} - - - {trailing} - - {open ? ( - - {pairs.map((pair, i) => ( - - ))} - {callCount > pairs.length ? ( - … {callCount - pairs.length} more tool calls - ) : null} - - ) : null} - - ) -} - /** Subtle top-right controls for an agent message: copy its prose, or scroll so * this message's top aligns to the top of the viewport. */ function AgentControls({ @@ -280,7 +102,13 @@ function MobileNativeChatMessageImpl({ fontScale = 1, messageIndex, onScrollToMessage, - onOpenFile + onOpenFile, + turnStatus, + turnExpanded, + turnKey, + onToggleTurn, + activeTurnIsWorking, + structuredActivityUi = false }: { message: NativeChatMessage toolsExpanded?: boolean @@ -291,6 +119,18 @@ function MobileNativeChatMessageImpl({ /** Ask the list to align this message's top to the top of the viewport. */ onScrollToMessage?: (index: number) => void onOpenFile?: (relativePath: string) => void + /** This turn's status row, rendered under a user message (desktop parity). */ + turnStatus?: NativeChatTurnStatus | null + /** Whether the turn caret has disclosed this turn's activity. */ + turnExpanded?: boolean + /** Set only when this row's turn has settled and can disclose its activity. */ + turnKey?: string + /** Stable across renders; the row supplies its own key when tapped. */ + onToggleTurn?: (turnKey: string) => void + /** Session-level working state for this message's turn; gates the live tool row. */ + activeTurnIsWorking?: boolean + /** Structured lane only: live tool progress plus the turn-status disclosure. */ + structuredActivityUi?: boolean }): React.JSX.Element { const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' @@ -310,6 +150,20 @@ function MobileNativeChatMessageImpl({ // tool calls fold into a collapsible run beneath. The user's own messages get // an inverted (filled accent) bubble so they stand apart from agent prose. const { prose, tools } = splitNativeChatBlocks(message.blocks) + const activeCall = structuredActivityUi + ? selectActiveToolCall(tools, { activeTurnIsWorking }) + : null + // A completed turn's activity belongs behind the turn-status caret. Leaving the + // grouped row visible made a failed child command read as a failed response. + // The composer's global Tools toggle still overrides this, or it would silently + // do nothing on every settled turn. + const settledToolsHidden = + structuredActivityUi && + activeCall == null && + activeTurnIsWorking === false && + !turnExpanded && + !toolsExpanded + const showToolRun = tools.length > 0 && !settledToolsHidden const handleCopy = (): void => { const text = nativeChatMessageText(message.blocks) @@ -338,39 +192,52 @@ function MobileNativeChatMessageImpl({ ) : null return ( - - - {prose.map((block, index) => ( - - ))} - {tools.length > 0 ? ( - - ) : controls ? ( - {controls} - ) : null} + <> + + + {prose.map((block, index) => ( + + ))} + {showToolRun ? ( + + ) : controls ? ( + {controls} + ) : null} + - + {turnStatus ? ( + onToggleTurn(turnKey) : undefined} + /> + ) : null} + ) } diff --git a/mobile/src/session/MobileNativeChatOverlay.tsx b/mobile/src/session/MobileNativeChatOverlay.tsx index 23724300ddf..357a089466e 100644 --- a/mobile/src/session/MobileNativeChatOverlay.tsx +++ b/mobile/src/session/MobileNativeChatOverlay.tsx @@ -71,6 +71,7 @@ export function MobileNativeChatOverlay({ error={session.error} agent={controller.nativeChatAgent} agentWorking={controller.nativeChatAgentWorking} + structuredActivityUi={controller.nativeChatStructured} streaming={streaming} onStop={controller.handleNativeChatStop} ask={controller.nativeChatAsk} diff --git a/mobile/src/session/MobileNativeChatPromptCard.tsx b/mobile/src/session/MobileNativeChatPromptCard.tsx new file mode 100644 index 00000000000..470ba2ee8b6 --- /dev/null +++ b/mobile/src/session/MobileNativeChatPromptCard.tsx @@ -0,0 +1,74 @@ +import type { AskAnswerSelection, AskPrompt } from '../../../src/shared/native-chat-ask' +import { MobileNativeChatAsk } from './MobileNativeChatAsk' +import { MobileNativeChatPermission } from './MobileNativeChatPermission' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import { MobileNativeChatQuestion } from './MobileNativeChatQuestion' +import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question' + +/** The one pending agent prompt shown above the composer: a structured + * AskUserQuestion wins, then a heuristic permission, then a heuristic question. + * The controller owns dismissal (it must survive this subtree unmounting on a + * view toggle); `ask` arrives already nulled while dismissed. */ +export function MobileNativeChatPromptCard({ + ask, + askKey, + onDismissAsk, + onAnswerAsk, + onCancelAsk, + permission, + onRespondPermission, + question, + onAnswerQuestion +}: { + ask?: AskPrompt | null + askKey?: string | null + onDismissAsk?: () => void + onAnswerAsk?: (prompt: AskPrompt, selections: AskAnswerSelection[]) => Promise + onCancelAsk?: () => Promise + permission?: MobileChatPermission | null + onRespondPermission?: (send: string) => Promise + question?: MobileChatQuestion | null + onAnswerQuestion?: (text: string) => Promise +}): React.JSX.Element | null { + if (ask) { + return ( + { + const accepted = (await onAnswerAsk?.(ask, selections)) ?? false + if (accepted) { + onDismissAsk?.() + } + return accepted + }} + onCancel={async () => { + const accepted = (await onCancelAsk?.()) ?? false + if (accepted) { + onDismissAsk?.() + } + return accepted + }} + /> + ) + } + if (permission) { + return ( + (await onRespondPermission?.(send)) ?? false} + /> + ) + } + if (question) { + return ( + (await onAnswerQuestion?.(text)) ?? false} + /> + ) + } + return null +} diff --git a/mobile/src/session/MobileNativeChatToolRun.tsx b/mobile/src/session/MobileNativeChatToolRun.tsx new file mode 100644 index 00000000000..db3ddbea063 --- /dev/null +++ b/mobile/src/session/MobileNativeChatToolRun.tsx @@ -0,0 +1,250 @@ +import { useEffect, useRef, useState } from 'react' +import { Animated, Pressable, Text, View } from 'react-native' +import { ChevronDown, SquareChevronRight, SquareTerminal, Wrench } from 'lucide-react-native' +import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff' +import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff' +import { pairToolBlocks } from '../../../src/shared/native-chat-tool-fold' +import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold' +import { + createToolInputDisplay, + summarizeToolRun, + truncateToolDetail +} from '../../../src/shared/native-chat-tool-summary' +import { + describeActiveToolCall, + formatActiveToolLabel, + formatToolCallCount, + isCommandToolName, + selectActiveToolCall +} from '../../../src/shared/native-chat-tool-activity' +import type { NativeChatBlock } from '../../../src/shared/native-chat-types' +import { colors } from '../theme/mobile-theme' +import { styles } from './mobile-native-chat-message-styles' + +const MAX_VISIBLE_TOOL_PAIRS = 6 +const MAX_TOOL_RUN_DIFF_ROWS = 240 + +function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element { + return ( + + {lines.map((line, i) => ( + + {line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '} + {line.text} + + ))} + + ) +} + +/** A single inline tool line — `▸ ToolName preview` — that expands in place to + * show the call's diff/input or the result's body. Mirrors the reference design + * where tool calls read as flat lines in the conversation, not boxed blocks. */ +function ResultBody({ + output, + isError, + diff +}: { + output: string + isError?: boolean + diff: DiffLine[] | null +}): React.JSX.Element { + if (diff) { + return + } + return ( + + {truncateToolDetail(output)} + + ) +} + +/** One request: a tool call and its result rendered together as a single + * expandable line. `defaultExpanded` lets the group toggle open every line. */ +function ToolLine({ + pair, + defaultExpanded, + diffLineLimit, + onOpenFile +}: { + pair: ToolPair + defaultExpanded: boolean + diffLineLimit: number + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const [expanded, setExpanded] = useState(defaultExpanded) + const { call, result } = pair + const name = call ? call.name : 'Result' + const inputDisplay = call ? createToolInputDisplay(call.input) : null + const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? '' + // Why: collapsed tool rows are the common path; defer bounded diff parsing + // and detail formatting until the user asks to reveal the detail. + const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null + const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null + const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined + const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true + // The group toggle opens every line at once, bypassing the tap guard, so the + // panel has to consult it too — else a detail-less row echoes its own label + // under itself and no tap can dismiss it. + const showDetail = hasDetail && expanded + // A tool that targets a file (Read/Edit/Write…) renders its preview as a + // tappable link that opens the file, independent of the line's expand tap. + const filePath = inputDisplay?.filePath ?? null + const openable = filePath !== null && onOpenFile !== undefined + return ( + + hasDetail && setExpanded((v) => !v)} + hitSlop={6} + > + {showDetail ? ( + + ) : ( + + )} + {name} + {preview ? ( + onOpenFile!(filePath!) : undefined} + suppressHighlighting={!openable} + > + {preview} + + ) : null} + + {showDetail ? ( + + {callDiff ? : null} + {callDetail ? {callDetail} : null} + {result ? ( + + ) : null} + + ) : null} + + ) +} + +/** Breathing label for a still-running tool, matching desktop's `animate-pulse`. */ +function PulsingText({ + style, + numberOfLines, + children +}: { + style?: React.ComponentProps['style'] + numberOfLines?: number + children: React.ReactNode +}): React.JSX.Element { + const pulse = useRef(new Animated.Value(1)).current + useEffect(() => { + const animation = Animated.loop( + Animated.sequence([ + Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), + Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) + ]) + ) + animation.start() + return () => animation.stop() + }, [pulse]) + return ( + + {children} + + ) +} + +/** A run of a message's tool calls/results, collapsed to a one-line summary that + * expands to the individual inline tool lines. `defaultExpanded` lets the global + * toolbar toggle drive every run at once while still allowing per-run override. */ +export function ToolRun({ + blocks, + defaultExpanded, + expandChildren, + activeCall, + trailing, + onOpenFile +}: { + blocks: NativeChatBlock[] + defaultExpanded: boolean + /** Child tool lines stay collapsed when the turn caret drove the run open. */ + expandChildren: boolean + /** The still-running call, when the turn is live (desktop parity). */ + activeCall: ReturnType + trailing?: React.ReactNode + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const [open, setOpen] = useState(defaultExpanded) + const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS) + const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1))) + let callCount = 0 + for (const block of blocks) { + if (block.type === 'tool-call') { + callCount++ + } + } + callCount ||= pairs.length + const summary = summarizeToolRun(blocks) + const ActiveToolIcon = activeCall && isCommandToolName(activeCall.name) ? SquareTerminal : Wrench + return ( + + + {activeCall ? ( + setOpen((v) => !v)} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded: open }} + accessibilityLiveRegion="polite" + > + + + {formatActiveToolLabel(describeActiveToolCall(activeCall))} + + {open ? : null} + + ) : ( + setOpen((v) => !v)} hitSlop={6}> + {open ? ( + + ) : ( + + )} + {callCount}× + + {summary || formatToolCallCount(callCount)} + + + )} + {trailing} + + {open ? ( + + {pairs.map((pair, i) => ( + + ))} + {callCount > pairs.length ? ( + … {callCount - pairs.length} more tool calls + ) : null} + + ) : null} + + ) +} diff --git a/mobile/src/session/MobileNativeChatTurnStatus.test.ts b/mobile/src/session/MobileNativeChatTurnStatus.test.ts new file mode 100644 index 00000000000..78ac01e0d37 --- /dev/null +++ b/mobile/src/session/MobileNativeChatTurnStatus.test.ts @@ -0,0 +1,112 @@ +import { createElement } from 'react' +import { act, create, type ReactTestInstance, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('react-native', async () => { + const React = await import('react') + const Text = ({ children, ...props }: { children?: unknown }): unknown => + React.createElement('Text', props, children) + return { + Animated: { + Text, + Value: class { + constructor(private value: number) {} + setValue(next: number): void { + this.value = next + } + }, + loop: (animation: unknown) => animation, + sequence: () => ({ start: vi.fn(), stop: vi.fn() }), + timing: () => ({ start: vi.fn(), stop: vi.fn() }) + }, + Pressable: ({ children, ...props }: { children?: unknown }) => + React.createElement('Pressable', props, children), + Text, + View: ({ children, ...props }: { children?: unknown }) => + React.createElement('View', props, children), + StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } + } +}) +vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' })) + +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' + +describe('MobileNativeChatTurnStatus', () => { + let renderer: ReactTestRenderer | null = null + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-09-04T00:00:00Z')) + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + vi.useRealTimers() + }) + + function render(props: { + startedAt: number | null + thinking: boolean + workedSeconds?: number | null + expanded?: boolean + onToggleExpanded?: () => void + }): ReactTestRenderer { + act(() => { + renderer = create(createElement(MobileNativeChatTurnStatus, props)) + }) + return renderer! + } + + const labels = (node: ReactTestInstance): string[] => + node.findAllByType('Text' as never).map((text) => String(text.children.join(''))) + + it('reads "Thinking" before the turn produces output', () => { + const tree = render({ startedAt: Date.now(), thinking: true }) + expect(labels(tree.root)).toEqual(['Thinking']) + }) + + it('counts up once the turn is producing output', () => { + const startedAt = Date.now() + const tree = render({ startedAt, thinking: false }) + expect(labels(tree.root)).toEqual(['Working for 0s']) + act(() => { + vi.advanceTimersByTime(12_000) + }) + expect(labels(tree.root)).toEqual(['Working for 12s']) + }) + + it('settles to a tappable "Worked for" row that toggles the turn', () => { + const onToggleExpanded = vi.fn() + const tree = render({ + startedAt: Date.now(), + thinking: false, + workedSeconds: 184, + onToggleExpanded + }) + expect(labels(tree.root)).toEqual(['Worked for 3m 4s']) + const button = tree.root.findByType('Pressable' as never) + expect(button.props.accessibilityLabel).toBe('Toggle turn details') + expect(button.props.accessibilityState).toEqual({ expanded: false }) + act(() => button.props.onPress()) + expect(onToggleExpanded).toHaveBeenCalledOnce() + }) + + it('stays a plain row when the settled turn has nothing to disclose', () => { + const tree = render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + expect(tree.root.findAllByType('Pressable' as never)).toHaveLength(0) + expect(labels(tree.root)).toEqual(['Worked for 5s']) + }) + + it('holds no interval once the turn has settled', () => { + render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + expect(vi.getTimerCount()).toBe(0) + }) + + it('announces the live row to assistive tech', () => { + const tree = render({ startedAt: Date.now(), thinking: true }) + const row = tree.root.findByType('View' as never) + expect(row.props.accessibilityLiveRegion).toBe('polite') + expect(row.props.accessibilityLabel).toBe('Agent is responding') + }) +}) diff --git a/mobile/src/session/MobileNativeChatTurnStatus.tsx b/mobile/src/session/MobileNativeChatTurnStatus.tsx new file mode 100644 index 00000000000..4ce73cdcd38 --- /dev/null +++ b/mobile/src/session/MobileNativeChatTurnStatus.tsx @@ -0,0 +1,117 @@ +import { useEffect, useRef, useState } from 'react' +import { Animated, Pressable, StyleSheet, Text, View } from 'react-native' +import { ChevronRight } from 'lucide-react-native' +import { + formatNativeChatTurnStatusLabel, + NATIVE_CHAT_TURN_STATUS_COPY, + nativeChatElapsedSeconds +} from '../../../src/shared/native-chat-turn-status' +import { colors, spacing, typography } from '../theme/mobile-theme' + +/** Seconds tick only while a turn is actually counting, so a settled transcript + * holds no timers. */ +function useElapsedSeconds(startedAt: number | null, counting: boolean): number { + // Preserves the pre-stamp epoch for the frame before the turn's startedAt lands. + const [mountedAt] = useState(() => Date.now()) + const [now, setNow] = useState(() => Date.now()) + useEffect(() => { + if (!counting) { + return + } + setNow(Date.now()) + const timer = setInterval(() => setNow(Date.now()), 1_000) + return () => clearInterval(timer) + }, [counting]) + return counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 +} + +/** The per-turn status row — "Thinking", then "Working for 12s" while the turn + * runs, settling to a tappable "Worked for 3m 4s" that discloses the turn's + * tool activity. Desktop parity: `NativeChatWorkingStatus`. */ +export function MobileNativeChatTurnStatus({ + startedAt, + thinking, + workedSeconds, + expanded = false, + onToggleExpanded +}: { + startedAt: number | null + thinking: boolean + workedSeconds?: number | null + expanded?: boolean + onToggleExpanded?: () => void +}): React.JSX.Element { + const counting = !thinking && workedSeconds == null + const elapsedSeconds = useElapsedSeconds(startedAt, counting) + const label = formatNativeChatTurnStatusLabel({ thinking, workedSeconds, elapsedSeconds }) + + const pulse = useRef(new Animated.Value(1)).current + useEffect(() => { + if (!thinking) { + pulse.setValue(1) + return + } + const animation = Animated.loop( + Animated.sequence([ + Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), + Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) + ]) + ) + animation.start() + return () => animation.stop() + }, [pulse, thinking]) + + const rowStyle = [styles.row, thinking ? null : styles.rowSettled] + + if (workedSeconds != null && onToggleExpanded) { + return ( + [...rowStyle, pressed && styles.pressed]} + onPress={onToggleExpanded} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded }} + accessibilityLabel={NATIVE_CHAT_TURN_STATUS_COPY.toggleDetails} + > + {label} + + + + + ) + } + + return ( + + {label} + + ) +} + +const styles = StyleSheet.create({ + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.xs, + minHeight: 28, + paddingHorizontal: spacing.md + }, + rowSettled: { + borderBottomWidth: StyleSheet.hairlineWidth, + borderBottomColor: colors.borderSubtle + }, + pressed: { + opacity: 0.6 + }, + label: { + color: colors.textMuted, + fontSize: typography.bodySize + }, + caretOpen: { + transform: [{ rotate: '90deg' }] + } +}) diff --git a/mobile/src/session/MobileNativeChatView.test.ts b/mobile/src/session/MobileNativeChatView.test.ts index 9d171e3ac8a..d101c3f0ef6 100644 --- a/mobile/src/session/MobileNativeChatView.test.ts +++ b/mobile/src/session/MobileNativeChatView.test.ts @@ -72,6 +72,9 @@ type Overrides = { inputLockReason?: 'disconnected' | 'waiting' | null onSend?: (text: string) => Promise pending?: Parameters[0]['pending'] + structuredActivityUi?: boolean + agentWorking?: boolean + sendSurfaceId?: string } function assistantTurn(id: string, text: string): NativeChatMessage { @@ -243,4 +246,111 @@ describe('MobileNativeChatView', () => { vi.useRealTimers() } }) + + describe('structured turn status wiring', () => { + const userTurn = (id: string, text: string): NativeChatMessage => ({ + id, + role: 'user', + blocks: [{ type: 'text', text }], + timestamp: 0, + source: 'transcript' + }) + + function rowProps(id: string): Record { + return (renderedRow(id) as { props: Record }).props + } + + function workingIndicators(): ReactTestInstance[] { + return renderer!.root.findAll((node) => node.type === 'WorkingIndicator') + } + + it('gives the live user turn a status row and drops the three-dot indicator', async () => { + const folded = [userTurn('u1', 'go')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + const props = rowProps('u1') + expect(props.structuredActivityUi).toBe(true) + expect(props.turnStatus).toMatchObject({ thinking: true, workedSeconds: null }) + expect(props.activeTurnIsWorking).toBe(true) + expect(workingIndicators()).toHaveLength(0) + }) + + it('keeps the bridge lane on the three-dot indicator with no turn status', async () => { + const folded = [userTurn('u1', 'go')] + await render({ messages: folded, folded, agentWorking: true }) + const props = rowProps('u1') + expect(props.structuredActivityUi).toBe(false) + expect(props.turnStatus).toBeNull() + expect(props.activeTurnIsWorking).toBe(false) + expect(workingIndicators()).toHaveLength(1) + }) + + it('settles the finished turn to a tappable duration', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + expect(rowProps('u1').turnStatus).toMatchObject({ thinking: false, workedSeconds: null }) + await update({ messages: folded, folded, structuredActivityUi: true, agentWorking: false }) + const settled = rowProps('u1') + expect(settled.turnStatus).toMatchObject({ thinking: false }) + expect((settled.turnStatus as { workedSeconds: number | null }).workedSeconds).toBeTypeOf( + 'number' + ) + expect(settled.onToggleTurn).toBeTypeOf('function') + expect(settled.activeTurnIsWorking).toBe(false) + }) + + it('hangs no status row on an assistant row', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + expect(rowProps('a1').turnStatus).toBeNull() + // The assistant row still belongs to the live turn, so its tool row stays visible. + expect(rowProps('a1').activeTurnIsWorking).toBe(true) + }) + + it('does not carry a running turn clock across chat surfaces', async () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const firstTab = [userTurn('u1', 'first')] + await render({ + messages: firstTab, + folded: firstTab, + structuredActivityUi: true, + agentWorking: true, + sendSurfaceId: 'host\0worktree\0tab-a' + }) + expect(rowProps('u1').turnStatus).toMatchObject({ startedAt: 1_000 }) + + vi.setSystemTime(12_000) + const secondTab = [userTurn('u2', 'second')] + await update({ + messages: secondTab, + folded: secondTab, + structuredActivityUi: true, + agentWorking: true, + sendSurfaceId: 'host\0worktree\0tab-b' + }) + + expect(rowProps('u2').turnStatus).toMatchObject({ startedAt: 12_000 }) + } finally { + vi.useRealTimers() + } + }) + + it('does not treat pre-user history as part of the live turn', async () => { + const history = [ + assistantTurn('a0', 'before the first prompt'), + userTurn('u1', 'go'), + assistantTurn('a1', 'working') + ] + await render({ + messages: history, + folded: history, + structuredActivityUi: true, + agentWorking: true + }) + + expect(rowProps('a0').activeTurnIsWorking).toBe(false) + expect(rowProps('a1').activeTurnIsWorking).toBe(true) + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatView.tsx b/mobile/src/session/MobileNativeChatView.tsx index 4db4e437b43..59c024435b6 100644 --- a/mobile/src/session/MobileNativeChatView.tsx +++ b/mobile/src/session/MobileNativeChatView.tsx @@ -21,16 +21,16 @@ import { type MobileNativeChatPendingItem } from './mobile-native-chat-render-data' import { useMobileNativeChatPinchGesture } from './use-mobile-native-chat-pinch-gesture' +import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' import { MobileAgentWorkingIndicator } from './MobileAgentWorkingIndicator' import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment' import { MobileNativeChatComposer } from './MobileNativeChatComposer' +import { MobileNativeChatPromptCard } from './MobileNativeChatPromptCard' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import type { MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatSessionOptionPickersProps } from './MobileNativeChatSessionOptionPickers' import { MobileNativeChatMessage } from './MobileNativeChatMessage' -import { MobileNativeChatAsk } from './MobileNativeChatAsk' -import { MobileNativeChatPermission } from './MobileNativeChatPermission' -import type { MobileChatPermission } from './mobile-native-chat-permission' -import { MobileNativeChatQuestion } from './MobileNativeChatQuestion' -import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatStatus } from './use-mobile-native-chat-session' const INPUT_LOCK_SETTLE_MS = 600 @@ -49,6 +49,9 @@ type Props = { /** Resolved agent for this chat; names the empty-state copy (desktop parity). */ agent?: string | null agentWorking?: boolean + /** Structured lane: per-turn "Working for N" status plus live tool progress, + * replacing the bridge lane's static three-dot working row (desktop parity). */ + structuredActivityUi?: boolean /** Interrupt the agent mid-turn (shown as a Stop button on the working bar). */ onStop?: () => void /** Live partial assistant text to show as an in-progress bubble, already gated @@ -126,6 +129,7 @@ export function MobileNativeChatView({ error, agent, agentWorking, + structuredActivityUi = false, onStop, streaming, hasMore, @@ -252,6 +256,15 @@ export function MobileNativeChatView({ listRef.current?.scrollToIndex({ index, viewPosition: 0, animated: true }) }, []) + // Per-turn "Thinking / Working for N / Worked for N" rows. The structured lane + // owns them; the bridge lane keeps its three-dot indicator. + const turns = useMobileNativeChatTurnDisclosure({ + messages: data, + enabled: structuredActivityUi, + isWorking: agentWorking === true, + scopeKey: sendSurfaceId + }) + const renderItem = useCallback( ({ item, index }: { item: NativeChatMessage; index: number }) => ( ), - [toolsExpanded, fontScale, onScrollToMessage, onOpenFile] + [toolsExpanded, fontScale, onScrollToMessage, onOpenFile, structuredActivityUi, turns] ) const emptyState = mobileNativeChatEmptyState(status, agent ?? null, error) @@ -337,6 +353,15 @@ export function MobileNativeChatView({ ) : null } + ListFooterComponent={ + turns.activeTurnIsUnanchored && turns.active ? ( + + ) : null + } ListEmptyComponent={ emptyState ? ( @@ -360,47 +385,22 @@ export function MobileNativeChatView({ ) : null} )} - {/* Pending agent prompt: a structured AskUserQuestion wins, then a - heuristic permission, then a heuristic question. The controller owns - dismissal (it must survive this subtree unmounting on a view toggle); - `ask` arrives already nulled while dismissed. */} - {ask ? ( - { - const accepted = (await onAnswerAsk?.(ask, selections)) ?? false - if (accepted) { - onDismissAsk?.() - } - return accepted - }} - onCancel={async () => { - const accepted = (await onCancelAsk?.()) ?? false - if (accepted) { - onDismissAsk?.() - } - return accepted - }} - /> - ) : permission ? ( - (await onRespondPermission?.(send)) ?? false} - /> - ) : question ? ( - (await onAnswerQuestion?.(text)) ?? false} - /> - ) : null} + {/* Chrome row above the composer: the working indicator and the global tool-calls expand/collapse toggle on the left, Stop in the far corner. */} - {agentWorking ? : null} + {agentWorking && !structuredActivityUi ? : null} [styles.chromeToggle, pressed && styles.pressed]} onPress={() => setToolsExpanded((v) => !v)} diff --git a/mobile/src/session/mobile-native-chat-controller-contract.ts b/mobile/src/session/mobile-native-chat-controller-contract.ts index 890a3a1562e..53187e0d6db 100644 --- a/mobile/src/session/mobile-native-chat-controller-contract.ts +++ b/mobile/src/session/mobile-native-chat-controller-contract.ts @@ -25,6 +25,8 @@ export type MobileNativeChatController = { chatPending: MobileNativeChatPendingMessage[] chatImagePreviewsByMessageId: Record nativeChatSession: ReturnType + /** Structured lane: drives the per-turn status row and live tool progress. */ + nativeChatStructured: boolean nativeChatAgentWorking: boolean nativeChatStreamingText?: string /** Agent mid-turn, regardless of whether chat is the visible view. */ diff --git a/mobile/src/session/mobile-native-chat-message-styles.ts b/mobile/src/session/mobile-native-chat-message-styles.ts index ad7cf4b4009..7ae1128445a 100644 --- a/mobile/src/session/mobile-native-chat-message-styles.ts +++ b/mobile/src/session/mobile-native-chat-message-styles.ts @@ -80,6 +80,18 @@ export const styles = StyleSheet.create({ fontFamily: typography.monoFamily, fontSize: MONO_SIZE }, + toolRunActive: { + flex: 1, + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm, + paddingVertical: 3 + }, + toolRunActiveLabel: { + flex: 1, + color: colors.textSecondary, + fontSize: typography.bodySize + }, toolRunBody: { paddingLeft: spacing.sm, borderLeftWidth: 2, diff --git a/mobile/src/session/use-mobile-native-chat-controller.ts b/mobile/src/session/use-mobile-native-chat-controller.ts index bf944398e87..a946956f8d6 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.ts @@ -10,9 +10,8 @@ import { useMobileNativeChatDrafts } from './use-mobile-native-chat-drafts' import { useMobileNativeChatFileSearch } from './use-mobile-native-chat-file-search' import { useMobileNativeChatMessageSend } from './use-mobile-native-chat-message-send' import { mobileNativeChatStreamPreview } from './mobile-native-chat-streaming-gate' -import { useMobileNativeChatSession } from './use-mobile-native-chat-session' import { useMobileNativeChatSessionOptionController } from './use-mobile-native-chat-session-option-controller' -import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' +import { useMobileNativeChatSessionLane } from './use-mobile-native-chat-session-lane' import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge' import { useMobileNativeChatPrompts } from './use-mobile-native-chat-prompts' import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' @@ -82,27 +81,19 @@ export function useMobileNativeChatController(args: { nativeChatTranscriptIsLocalReadable }) - const legacyNativeChatSession = useMobileNativeChatSession({ - client, - sourceIdentity, - agent: activeChatStructured ? null : (activeChatResolution?.agent ?? null), - sessionId: activeChatStructured ? null : activeChatSessionId, - transcriptPath: activeChatStructured ? null : (activeChatResolution?.transcriptPath ?? null) - }) - const structuredNativeChat = useMobileStructuredAgentSession({ - client, - sessionId: activeChatStructured ? activeChatSessionId : null, - sourceIdentity, - enabled: showNativeChat, - // Holds are connection-scoped; dropping this on transport loss lets the hook - // reacquire the provider without clearing the cached transcript. - connected: connState === 'connected', - agent: activeChatStructured ? activeChatAgent : null, - onSendError - }) - const nativeChatSession = activeChatStructured - ? structuredNativeChat.session - : legacyNativeChatSession + const { structuredSession: structuredNativeChat, session: nativeChatSession } = + useMobileNativeChatSessionLane({ + client, + structured: activeChatStructured, + agent: activeChatAgent, + resolvedAgent: activeChatResolution?.agent ?? null, + transcriptPath: activeChatResolution?.transcriptPath ?? null, + sessionId: activeChatSessionId, + sourceIdentity, + enabled: showNativeChat, + connState, + onSendError + }) const { composerText: chatComposerText, setComposerText: setChatComposerText, @@ -303,6 +294,8 @@ export function useMobileNativeChatController(args: { chatPending, chatImagePreviewsByMessageId, nativeChatSession, + /** Structured lane: drives the per-turn status row and live tool progress. */ + nativeChatStructured: activeChatStructured, nativeChatAgentWorking, nativeChatStreamingText, nativeChatStreamLive, diff --git a/mobile/src/session/use-mobile-native-chat-session-lane.ts b/mobile/src/session/use-mobile-native-chat-session-lane.ts new file mode 100644 index 00000000000..fc465de902b --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-session-lane.ts @@ -0,0 +1,59 @@ +import type { RpcClient } from '../transport/rpc-client' +import type { ConnectionState } from '../transport/types' +import { useMobileNativeChatSession } from './use-mobile-native-chat-session' +import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' + +/** Mounts both transcript sources and hands back the one this tab's lane owns. + * Both hooks always run (hook order is fixed); the inactive lane is starved of + * its identity inputs rather than unmounted, so a lane flip keeps its cache. */ +export function useMobileNativeChatSessionLane({ + client, + structured, + agent, + resolvedAgent, + transcriptPath, + sessionId, + sourceIdentity, + enabled, + connState, + onSendError +}: { + client: RpcClient | null + structured: boolean + /** Agent id for the structured provider session. */ + agent: string | null + /** Agent resolved from the terminal, for the bridge transcript reader. */ + resolvedAgent: string | null + transcriptPath: string | null + sessionId: string | null + sourceIdentity: Parameters[0]['sourceIdentity'] + enabled: boolean + connState: ConnectionState + onSendError: (message: string) => void +}): { + structuredSession: ReturnType + session: ReturnType +} { + const bridgeSession = useMobileNativeChatSession({ + client, + sourceIdentity, + agent: structured ? null : resolvedAgent, + sessionId: structured ? null : sessionId, + transcriptPath: structured ? null : transcriptPath + }) + const structuredSession = useMobileStructuredAgentSession({ + client, + sessionId: structured ? sessionId : null, + sourceIdentity, + enabled, + // Holds are connection-scoped; dropping this on transport loss lets the hook + // reacquire the provider without clearing the cached transcript. + connected: connState === 'connected', + agent: structured ? agent : null, + onSendError + }) + return { + structuredSession, + session: structured ? structuredSession.session : bridgeSession + } +} diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx new file mode 100644 index 00000000000..8467684ce16 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx @@ -0,0 +1,153 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' + +function userMessage(id: string): NativeChatMessage { + return { + id, + role: 'user', + blocks: [{ type: 'text', text: id }], + timestamp: null, + source: 'transcript' + } +} + +function Harness({ + messages, + enabled, + isWorking = true, + scopeKey = 'host\0worktree\0tab-a' +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking?: boolean + scopeKey?: string +}): React.JSX.Element { + const disclosure = useMobileNativeChatTurnDisclosure({ + messages, + enabled, + isWorking, + scopeKey + }) + return createElement('result', { disclosure }) +} + +describe('useMobileNativeChatTurnDisclosure', () => { + let renderer: ReactTestRenderer | null = null + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + }) + + it('does not scan bridge-lane transcripts', () => { + const messages: NativeChatMessage[] = [ + { + id: 'u1', + role: 'user', + blocks: [{ type: 'text', text: 'go' }], + timestamp: null, + source: 'transcript' + } + ] + const findLastIndex = vi.spyOn(messages, 'findLastIndex') + const slice = vi.spyOn(messages, 'slice') + const filter = vi.spyOn(messages, 'filter') + const map = vi.spyOn(messages, 'map') + + act(() => { + renderer = create(createElement(Harness, { messages, enabled: false })) + }) + + expect(findLastIndex).not.toHaveBeenCalled() + expect(slice).not.toHaveBeenCalled() + expect(filter).not.toHaveBeenCalled() + expect(map).not.toHaveBeenCalled() + }) + + it('keeps a settled turn handler stable for NUL-delimited scope keys', () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const messages: NativeChatMessage[] = [ + { + id: 'u1', + role: 'user', + blocks: [{ type: 'text', text: 'go' }], + timestamp: null, + source: 'transcript' + } + ] + act(() => { + renderer = create(createElement(Harness, { messages, enabled: true })) + }) + vi.setSystemTime(6_000) + act(() => { + renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false })) + }) + const first = renderer!.root.findByType('result').props.disclosure.resolveRow(0, messages[0]) + + const refreshed = [...messages] + act(() => { + renderer?.update( + createElement(Harness, { messages: refreshed, enabled: true, isWorking: false }) + ) + }) + const second = renderer!.root + .findByType('result') + .props.disclosure.resolveRow(0, refreshed[0]) + + // The row carries the key; the handler itself lives on the hook and stays + // stable for the scope, so a re-render never disturbs a row's memo. + expect(first.turnKey).toBe('u1') + expect(second.turnKey).toBe('u1') + const firstHandler = renderer!.root.findByType('result').props.disclosure.onToggleTurn + expect(firstHandler).toBeTypeOf('function') + act(() => { + renderer?.update( + createElement(Harness, { messages: [...refreshed], enabled: true, isWorking: false }) + ) + }) + expect(renderer!.root.findByType('result').props.disclosure.onToggleTurn).toBe(firstHandler) + } finally { + vi.useRealTimers() + } + }) + + it('keeps at most the latest 128 turns expanded', () => { + vi.useFakeTimers() + try { + let messages: NativeChatMessage[] = [] + for (let index = 0; index < 129; index++) { + messages = messages.concat(userMessage(`u${index}`)) + vi.setSystemTime(index * 2_000) + act(() => { + if (renderer) { + renderer.update(createElement(Harness, { messages, enabled: true })) + } else { + renderer = create(createElement(Harness, { messages, enabled: true })) + } + }) + vi.setSystemTime(index * 2_000 + 1_000) + act(() => { + renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false })) + }) + const disclosureNow = renderer!.root.findByType('result').props.disclosure + const row = disclosureNow.resolveRow(index, messages[index]) + act(() => disclosureNow.onToggleTurn(row.turnKey)) + } + + const disclosure = renderer!.root.findByType('result').props.disclosure + const expanded = messages.filter( + (message, index) => disclosure.resolveRow(index, message).turnExpanded + ) + expect(expanded).toHaveLength(128) + expect(disclosure.resolveRow(0, messages[0]).turnExpanded).toBe(false) + expect(disclosure.resolveRow(128, messages[128]).turnExpanded).toBe(true) + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts new file mode 100644 index 00000000000..46b58f29cba --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts @@ -0,0 +1,126 @@ +import { useCallback, useMemo, useState } from 'react' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { + MOBILE_UNANCHORED_TURN_KEY, + useMobileNativeChatTurnStatus, + type NativeChatTurnStatus +} from './use-mobile-native-chat-turn-status' + +const EMPTY_TURN_IDS: ReadonlySet = new Set() +const EMPTY_TURN_KEYS: readonly undefined[] = [] +const MAX_EXPANDED_TURNS = 128 + +export type MobileNativeChatTurnRow = { + turnStatus: NativeChatTurnStatus | null + turnExpanded: boolean + /** Set only on a settled turn — the one row that has activity to disclose. */ + turnKey?: string + activeTurnIsWorking: boolean +} + +/** Owns the transcript's per-turn status rows and their disclosure state, and + * resolves what one list row needs. Bridge-lane chats pass `enabled: false` and + * keep their single three-dot working indicator instead. */ +export function useMobileNativeChatTurnDisclosure({ + messages, + enabled, + isWorking, + scopeKey +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking: boolean + /** Host/worktree/tab identity for timing and disclosure isolation. */ + scopeKey: string +}): { + active: NativeChatTurnStatus | null + /** True when the live turn has no user message to hang its status row under. */ + activeTurnIsUnanchored: boolean + onToggleTurn: (turnKey: string) => void + resolveRow: (index: number, message: NativeChatMessage) => MobileNativeChatTurnRow +} { + const turnStatuses = useMobileNativeChatTurnStatus({ + messages, + enabled, + isWorking, + scopeKey + }) + const [expandedTurns, setExpandedTurns] = useState<{ + scopeKey: string + turnIds: ReadonlySet + }>(() => ({ scopeKey, turnIds: new Set() })) + const expandedTurnIds = + expandedTurns.scopeKey === scopeKey ? expandedTurns.turnIds : EMPTY_TURN_IDS + const toggleExpandedTurn = useCallback( + (turnKey: string) => { + setExpandedTurns((current) => { + const next = new Set(current.scopeKey === scopeKey ? current.turnIds : []) + if (!next.delete(turnKey)) { + if (next.size >= MAX_EXPANDED_TURNS) { + const oldest = next.values().next().value + if (oldest) { + next.delete(oldest) + } + } + next.add(turnKey) + } + return { scopeKey, turnIds: next } + }) + }, + [scopeKey] + ) + // Resolve each row's turn boundary once — a findLast per row is quadratic on a + // long transcript. + const turnKeys = useMemo(() => { + if (!enabled) { + return EMPTY_TURN_KEYS + } + let turnKey: string | undefined + return messages.map((message) => { + if (message.role === 'user') { + turnKey = message.id + } + return turnKey + }) + }, [enabled, messages]) + + const { active, activeTurnKey, completedByTurn } = turnStatuses + const resolveRow = useCallback( + (index: number, message: NativeChatMessage): MobileNativeChatTurnRow => { + const turnKey = turnKeys[index] + const turnStatus = + !enabled || message.role !== 'user' + ? null + : turnKey === activeTurnKey + ? active + : turnKey + ? (completedByTurn[turnKey] ?? null) + : null + return { + turnStatus, + turnExpanded: turnKey ? expandedTurnIds.has(turnKey) : false, + // Why: the key travels and the row calls one stable handler with it. A + // closure per row would be a new identity every render of a streaming + // transcript, defeating the row's memo; caching one per turn would mean + // writing a ref during render, which react-freeze can discard. + turnKey: turnKey && turnStatus?.workedSeconds != null ? turnKey : undefined, + // With no user boundary at all, the session's working state stays authoritative. + activeTurnIsWorking: + enabled && + isWorking && + (turnKey === activeTurnKey || + (turnKey === undefined && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY)) + } + }, + [turnKeys, enabled, activeTurnKey, active, completedByTurn, expandedTurnIds, isWorking] + ) + + return { + active, + /** Stable for a given chat scope, so it never disturbs a row's memo. */ + onToggleTurn: toggleExpandedTurn, + activeTurnIsUnanchored: + enabled && active != null && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY, + resolveRow + } +} diff --git a/mobile/src/session/use-mobile-native-chat-turn-status.ts b/mobile/src/session/use-mobile-native-chat-turn-status.ts new file mode 100644 index 00000000000..13afbe70c09 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-status.ts @@ -0,0 +1,105 @@ +import { useEffect, useMemo, useRef, useState } from 'react' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { + nativeChatTurnHasResponse, + reduceNativeChatTurnTiming, + selectNativeChatTurnStatuses, + type NativeChatTurnStatus, + type NativeChatTurnTimingByTurn +} from '../../../src/shared/native-chat-turn-status' + +export type { NativeChatTurnStatus } + +export const MOBILE_UNANCHORED_TURN_KEY = '__unanchored__' +const EMPTY_TURN_TIMING_BY_TURN: NativeChatTurnTimingByTurn = Object.freeze({}) + +type ScopedTurnTiming = { + scopeKey: string + timingByTurn: NativeChatTurnTimingByTurn +} + +/** Per-turn "Thinking / Working for N / Worked for N" timing, on the same shared + * state machine the desktop renderer uses so the two surfaces stamp turns alike. */ +export function useMobileNativeChatTurnStatus({ + messages, + enabled, + isWorking, + workingStartedAt, + scopeKey +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking: boolean + workingStartedAt?: number | null + /** Host/worktree/tab identity. Timings never carry across chat surfaces. */ + scopeKey: string +}): { + active: NativeChatTurnStatus | null + completedByTurn: Readonly> + activeTurnKey: string +} { + const latestUserIndex = enabled + ? messages.findLastIndex((message) => message.role === 'user') + : -1 + const hasCurrentTurnResponse = enabled && nativeChatTurnHasResponse(messages, latestUserIndex) + const latestUserId = latestUserIndex !== -1 ? (messages[latestUserIndex]?.id ?? null) : null + const activeTurnKey = latestUserId ?? MOBILE_UNANCHORED_TURN_KEY + const [scopedTiming, setScopedTiming] = useState(() => ({ + scopeKey, + timingByTurn: {} + })) + // Do not expose the previous surface's state during the render before the + // timing effect adopts the new scope, or scan it while this UI is disabled. + const timingByTurn = + enabled && scopedTiming.scopeKey === scopeKey + ? scopedTiming.timingByTurn + : EMPTY_TURN_TIMING_BY_TURN + // An accepted send renders as `pending-N` until the transcript echo lands under + // its real id. That is one turn under two keys, so the clock must survive the swap. + const previousActiveTurn = useRef<{ scopeKey: string; turnKey: string } | null>(null) + + useEffect(() => { + if (!enabled) { + return + } + const validTurnKeys = new Set( + messages.filter((message) => message.role === 'user').map((message) => message.id) + ) + const previousActiveTurnKey = + previousActiveTurn.current?.scopeKey === scopeKey + ? previousActiveTurn.current.turnKey + : undefined + previousActiveTurn.current = { scopeKey, turnKey: activeTurnKey } + setScopedTiming((current) => { + const currentTiming = + current.scopeKey === scopeKey ? current.timingByTurn : EMPTY_TURN_TIMING_BY_TURN + const nextTiming = reduceNativeChatTurnTiming(currentTiming, { + activeTurnKey, + previousActiveTurnKey, + validTurnKeys, + isWorking, + workingStartedAt, + now: Date.now() + }) + return current.scopeKey === scopeKey && nextTiming === currentTiming + ? current + : { scopeKey, timingByTurn: nextTiming } + }) + }, [activeTurnKey, enabled, isWorking, messages, scopeKey, workingStartedAt]) + + // Why: the selection rebuilds its status objects on every call, and a streaming + // turn re-renders ~20x/s. Without this, every settled turn's row gets fresh + // props each tick and the memoized message rows all re-render. + const turnIsWorking = enabled && isWorking + const statuses = useMemo( + () => + selectNativeChatTurnStatuses(timingByTurn, { + activeTurnKey, + isWorking: turnIsWorking, + workingStartedAt, + hasCurrentTurnResponse + }), + [timingByTurn, activeTurnKey, turnIsWorking, workingStartedAt, hasCurrentTurnResponse] + ) + return { ...statuses, activeTurnKey } +} diff --git a/src/main/runtime/rpc/mobile-socket-wiring.test.ts b/src/main/runtime/rpc/mobile-socket-wiring.test.ts index 14d5beb8cd0..5ce3325083c 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.test.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.test.ts @@ -190,6 +190,58 @@ describe('MobileSocketWiring', () => { expect(wiring.connectionCount).toBe(0) }) + it('lets the capability RPC write capabilities back onto the socket', () => { + // `runtime.clientCapabilities.update` stores the advertised set by assigning + // `authenticatedSocket.clientCapabilities`. A read-only socket makes that a + // TypeError, the RPC answers `runtime_error`, and every structured + // agent-session tab is then projected away from a capable phone. + const desktop = generateKeyPair() + const phone = generateKeyPair() + const ws = new FakeSocket() + const transport = new FakeTransport() + const onText = vi.fn() + const wiring = new MobileSocketWiring({ + deviceRegistry: registryFor('device-1', 'valid-token', 'mobile'), + e2eeKeypair: { + publicKey: desktop.publicKey, + secretKey: desktop.secretKey, + publicKeyB64: Buffer.from(desktop.publicKey).toString('base64') + }, + onText, + onBinary: vi.fn(), + onClose: vi.fn() + }) + wiring.attachTransport(transport) + + transport.receive( + ws, + JSON.stringify({ + type: 'e2ee_hello', + publicKeyB64: Buffer.from(phone.publicKey).toString('base64') + }) + ) + const sharedKey = deriveSharedKey(phone.secretKey, desktop.publicKey) + transport.receive( + ws, + encrypt(JSON.stringify({ type: 'e2ee_auth', deviceToken: 'valid-token' }), sharedKey) + ) + transport.receive(ws, encrypt('{"id":"rpc-1","method":"status.get"}', sharedKey)) + + const socket = onText.mock.calls[0]?.[0] + expect(socket).toBeDefined() + expect(socket.clientCapabilities).toEqual([]) + + expect(() => { + socket.clientCapabilities = ['agent-session.structured.v1'] + }).not.toThrow() + expect(socket.clientCapabilities).toEqual(['agent-session.structured.v1']) + + // Later requests on the same connection must see the updated set, so the + // channel is the single source of truth rather than a detached copy. + transport.receive(ws, encrypt('{"id":"rpc-2","method":"status.get"}', sharedKey)) + expect(onText.mock.calls[1]?.[0].clientCapabilities).toEqual(['agent-session.structured.v1']) + }) + it('closes an unknown-token socket even when reporting the failure throws', () => { const desktop = generateKeyPair() const phone = generateKeyPair() diff --git a/src/main/runtime/rpc/mobile-socket-wiring.ts b/src/main/runtime/rpc/mobile-socket-wiring.ts index 384f403e9de..2d536b15038 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.ts @@ -165,9 +165,16 @@ export class MobileSocketWiring { ws, connectionId, device, + // Why: the channel owns the set for the whole connection, so this reads + // through rather than snapshotting. It must also WRITE through — the + // capability RPC updates the socket, and a getter-only property makes + // that a TypeError, which strands a capable phone with no capabilities. get clientCapabilities() { return channel.clientCapabilities }, + set clientCapabilities(next: readonly RuntimeCapability[]) { + channel.clientCapabilities = next + }, transport: metadata } this.authenticatedSockets.set(ws, socket) diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx index f16d01331f8..14cb0163f50 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx @@ -14,47 +14,24 @@ import { summarizeToolRun, truncateToolDetail } from './native-chat-tool-summary' +import { + describeActiveToolCall, + isCommandToolName, + NATIVE_CHAT_TOOL_ACTIVITY_COPY, + selectActiveToolCall +} from '../../../../shared/native-chat-tool-activity' import { NativeChatDiffView } from './NativeChatDiffView' -const COMMAND_TOOL_NAMES = new Set([ - 'bash', - 'shell', - 'powershell', - 'terminal', - 'execute', - 'run_command', - 'run_shell_command', - 'shell_command', - 'exec_command', - 'run_terminal_cmd', - 'run_terminal_command' -]) - -function normalizedToolName(name: string): string { - return name.trim().toLowerCase() -} - function activeToolLabel(call: Extract): string { - const preview = createToolInputDisplay(call.input).label - if (COMMAND_TOOL_NAMES.has(normalizedToolName(call.name))) { - return preview - ? translate('components.native-chat.tool.runningPreview', 'Running {{preview}}', { - preview - }) - : translate('components.native-chat.tool.runningCommand', 'Running command') - } - return preview - ? translate( - 'components.native-chat.tool.runningNamedPreview', - 'Running {{toolName}} {{preview}}', - { - toolName: call.name, - preview - } - ) - : translate('components.native-chat.tool.runningNamed', 'Running {{toolName}}', { - toolName: call.name - }) + const { key, toolName, preview } = describeActiveToolCall(call) + const copy = NATIVE_CHAT_TOOL_ACTIVITY_COPY[key] + return key === 'runningPreview' + ? translate('components.native-chat.tool.runningPreview', copy, { preview }) + : key === 'runningCommand' + ? translate('components.native-chat.tool.runningCommand', copy) + : key === 'runningNamedPreview' + ? translate('components.native-chat.tool.runningNamedPreview', copy, { toolName, preview }) + : translate('components.native-chat.tool.runningNamed', copy, { toolName }) } /** A single inline tool line — `▸ ToolName preview` — that expands in place to @@ -176,27 +153,19 @@ export function NativeChatToolRun({ const callCount = countToolCalls(blocks) || blocks.length const summary = summarizeToolRun(blocks) - const calls = blocks.filter(isToolCallBlock) - const activeCalls = structuredActivityUi - ? calls.filter( - (call) => - (call.state === 'running' || (call.state == null && activeTurnIsWorking === true)) && - activeTurnIsWorking !== false - ) - : [] - const latestActiveCall = activeCalls.at(-1) + const latestActiveCall = structuredActivityUi + ? selectActiveToolCall(blocks, { activeTurnIsWorking }) + : null const isSettled = latestActiveCall == null // The turn caret opens the activity group, while each child tool remains // collapsed. The global expand toolbar still opens child details together. const expandToolLines = expandOverride === undefined ? open : false const ActiveToolIcon = - latestActiveCall && COMMAND_TOOL_NAMES.has(normalizedToolName(latestActiveCall.name)) - ? SquareTerminal - : Wrench + latestActiveCall && isCommandToolName(latestActiveCall.name) ? SquareTerminal : Wrench const fallbackLabel = callCount === 1 - ? translate('components.native-chat.tool.countOne', '1 tool call') - : translate('components.native-chat.tool.countN', '{{value0}} tool calls', { + ? translate('components.native-chat.tool.countOne', NATIVE_CHAT_TOOL_ACTIVITY_COPY.countOne) + : translate('components.native-chat.tool.countN', NATIVE_CHAT_TOOL_ACTIVITY_COPY.countN, { value0: callCount }) diff --git a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx index 21145e94c94..e6c38de83f5 100644 --- a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx @@ -2,21 +2,14 @@ import { useState } from 'react' import { ChevronRight } from 'lucide-react' import { translate } from '@/i18n/i18n' import { useNow } from '@/hooks/use-now' +import { + describeNativeChatTurnStatus, + formatNativeChatDuration, + NATIVE_CHAT_TURN_STATUS_COPY, + nativeChatElapsedSeconds +} from '../../../../shared/native-chat-turn-status' -/** Format turn time without exposing an ever-growing raw seconds count. */ -export function formatNativeChatDuration(seconds: number): string { - const totalSeconds = Number.isFinite(seconds) ? Math.max(0, Math.floor(seconds)) : 0 - if (totalSeconds < 60) { - return `${totalSeconds}s` - } - const minutes = Math.floor(totalSeconds / 60) - const remainingSeconds = totalSeconds % 60 - if (minutes < 60) { - return `${minutes}m ${remainingSeconds}s` - } - const hours = Math.floor(minutes / 60) - return `${hours}h ${minutes % 60}m ${remainingSeconds}s` -} +export { formatNativeChatDuration } export function NativeChatWorkingStatus({ startedAt, @@ -39,21 +32,29 @@ export function NativeChatWorkingStatus({ // Why: preserves the old effect's `startedAt ?? Date.now()` epoch for the // single frame before the turn's startedAt lands. const [mountedAt] = useState(() => Date.now()) - const elapsedSeconds = counting - ? Math.max(0, Math.floor((now - (startedAt ?? mountedAt)) / 1000)) - : 0 + const elapsedSeconds = counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 + const { key, duration } = describeNativeChatTurnStatus({ + thinking, + workedSeconds, + elapsedSeconds + }) const label = - workedSeconds != null - ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}}', { - value0: formatNativeChatDuration(workedSeconds) - }) - : thinking - ? translate('components.native-chat.status.thinking', 'Thinking') - : translate('components.native-chat.status.workingFor', 'Working for {{value0}}', { - value0: formatNativeChatDuration(elapsedSeconds) - }) - + key === 'workedFor' + ? translate( + 'components.native-chat.status.workedFor', + NATIVE_CHAT_TURN_STATUS_COPY.workedFor, + { + value0: duration + } + ) + : key === 'thinking' + ? translate('components.native-chat.status.thinking', NATIVE_CHAT_TURN_STATUS_COPY.thinking) + : translate( + 'components.native-chat.status.workingFor', + NATIVE_CHAT_TURN_STATUS_COPY.workingFor, + { value0: duration } + ) const className = `flex min-h-8 items-center gap-1 text-sm text-muted-foreground${thinking ? '' : ' border-b border-border'}` const caret = workedSeconds != null ? ( @@ -67,7 +68,10 @@ export function NativeChatWorkingStatus({ +
+ {file.oldPath ? ( + <> + + {baseName(file.oldPath)} + + → + + ) : null} + + {baseName(file.path)} + + + {file.truncated ? ( + // Beside the counts rather than under the rows: a collapsed card, and + // one clipped down to no rows at all, would otherwise say nothing. + + {translate('components.native-chat.tool.diffTruncated', 'Diff truncated')} + + ) : null} + +
+ {hasBody && expanded ? ( + // Focusable so the rows can be scrolled from the keyboard. +
+ {(() => { + const seen = new Map() + return file.lines.map((line) => { + const signature = `${line.kind}:${line.oldLineNumber}:${line.newLineNumber}:${line.text}` + const occurrence = seen.get(signature) ?? 0 + seen.set(signature, occurrence + 1) + return ( + + ) + }) + })()} +
+ ) : null} +
+ ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx index 965ed0ad176..050b3f7c84b 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx @@ -2,8 +2,8 @@ import '@testing-library/jest-dom/vitest' -import { cleanup, render, screen } from '@testing-library/react' -import { afterEach, describe, expect, it } from 'vitest' +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' import type { NativeChatBlock } from '../../../../shared/native-chat-types' import { projectStructuredItemToNativeChat } from '../../../../shared/structured-agent-session-projection' @@ -32,6 +32,9 @@ describe('NativeChatToolRun', () => { { type: 'tool-call', name: 'apply_patch', + // The patch lives on the call in this lane, so the provider's own + // completion is what says the edit landed. + state: 'completed', input: { changes: [ { @@ -46,8 +49,9 @@ describe('NativeChatToolRun', () => { const { container } = render() - expect(screen.getByText('+after')).toBeInTheDocument() - expect(screen.getByText('-before')).toBeInTheDocument() + expect(screen.getByText('after')).toBeInTheDocument() + expect(screen.getByText('before')).toBeInTheDocument() + expect(screen.getByText('Edited file')).toBeInTheDocument() expect(container.querySelector('pre')).toBeNull() }) @@ -78,18 +82,156 @@ describe('NativeChatToolRun', () => { ) - expect(screen.getByText('+after')).toHaveClass( - 'bg-emerald-500/10', - 'text-[var(--git-decoration-added)]' - ) - expect(screen.getByText('-before')).toHaveClass( - 'bg-rose-500/10', - 'text-[var(--git-decoration-deleted)]' - ) + // Row grounds come from the diff tokens, not a hardcoded palette value. + expect(screen.getByText('after').closest('div')).toHaveClass('bg-[var(--diff-added-ground)]') + expect(screen.getByText('before').closest('div')).toHaveClass('bg-[var(--diff-removed-ground)]') expect(container).not.toHaveTextContent('"changes"') expect(container.querySelector('pre')).toBeNull() }) + it('keeps the provider error visible for an edit the agent could not apply', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'missing', new_string: 'now' } + }, + { type: 'tool-result', output: 'String to replace not found in file.', isError: true } + ] + + const { container } = render() + + expect(screen.queryByText('Edited file')).toBeNull() + const body = container.querySelector('pre') + expect(body).toHaveTextContent('String to replace not found in file.') + expect(body).toHaveClass('text-destructive') + }) + + it('leaves a `git diff` command as a command row rather than an edit card', () => { + const blocks: NativeChatBlock[] = [ + { type: 'tool-call', name: 'exec', input: { command: 'git diff' }, state: 'completed' }, + { + type: 'tool-result', + output: 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1 +1 @@\n-was\n+now' + } + ] + + const { container } = render() + + expect(screen.queryByText('Edited file')).toBeNull() + expect(container).toHaveTextContent('git diff') + }) + + it('shows no gutter number for a snippet edit, which cannot locate itself', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' }, + state: 'completed' + }, + { type: 'tool-result', output: 'ok' } + ] + + render() + + // Exact, because a snippet-relative number would sit ahead of the marker. + expect(screen.getByText('now').closest('div')?.textContent).toBe('+now') + expect(screen.getByText('was').closest('div')?.textContent).toBe('-was') + }) + + it('separates two regions of a file so the gutter jump is accounted for', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + state: 'completed' + }, + { + type: 'tool-result', + output: 'ok', + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 42, oldLines: 1, newStart: 42, newLines: 1, lines: ['-was', '+now'] }, + { oldStart: 310, oldLines: 1, newStart: 310, newLines: 1, lines: ['-old', '+new'] } + ] + } + } + ] + + render() + + const separators = screen.getAllByRole('separator') + expect(separators).toHaveLength(1) + expect(separators[0]).toHaveAccessibleName('Lines not shown') + }) + + it('offers no empty body for a delete, which names the file and nothing else', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'apply_patch', + input: { input: '*** Begin Patch\n*** Delete File: gone.ts\n*** End Patch' }, + state: 'completed' + } + ] + + render() + + expect(screen.getByTitle('gone.ts')).toBeInTheDocument() + // The header states the change; there is no body behind a disclosure. + expect(screen.getByText('Deleted file').closest('button')).not.toHaveAttribute('aria-expanded') + }) + + it('says a diff was clipped even while the card is collapsed', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Diff', + input: { path: 'src/a.ts' }, + state: 'completed' + }, + { type: 'tool-result', output: '@@ -1,3 +1,3 @@\n ctx\n-was\n+now\n… (48210 bytes)' } + ] + + // A defined expandOverride opens the run while leaving each card closed. + render() + + expect(screen.getByText('Diff truncated')).toBeInTheDocument() + expect(screen.queryByText('was')).toBeNull() + }) + + it('copies the diff as signed rows, with the region breaks left out', () => { + const writeClipboardText = vi.fn() + Object.assign(window, { api: { ui: { writeClipboardText } } }) + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + state: 'completed' + }, + { + type: 'tool-result', + output: 'ok', + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 1, oldLines: 2, newStart: 1, newLines: 2, lines: [' ctx', '-was', '+now'] }, + { oldStart: 90, oldLines: 1, newStart: 90, newLines: 1, lines: ['+tail'] } + ] + } + } + ] + + render() + fireEvent.click(screen.getByRole('button', { name: 'Copy diff' })) + + expect(writeClipboardText).toHaveBeenCalledWith(' ctx\n-was\n+now\n+tail') + }) + it('keeps a grouped active run to one stable row showing only the latest tool', () => { const blocks: NativeChatBlock[] = [ { type: 'tool-call', name: 'shell', input: { command: 'date' }, state: 'completed' }, diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx index 14cb0163f50..e91177b375a 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx @@ -1,4 +1,4 @@ -import { useEffect, useState } from 'react' +import { useEffect, useMemo, useState } from 'react' import { Check, ChevronRight, SquareTerminal, Wrench } from 'lucide-react' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' @@ -8,6 +8,13 @@ import { type NativeChatBlock } from '../../../../shared/native-chat-types' import { diffFromText, diffFromToolCall, type DiffLine } from './native-chat-diff' +import { NativeChatDiffCard } from './NativeChatDiffCard' +import { pairToolBlocks } from './native-chat-tool-fold' +import { + editFilesFromToolPair, + isEditToolName +} from '../../../../shared/native-chat-edit-normalize' +import type { NativeChatEditFile } from '../../../../shared/native-chat-edit-model' import { countToolCalls, createToolInputDisplay, @@ -128,6 +135,50 @@ function ToolLine({ ) } +type EditCardModel = { + editCards: Map + /** Result blocks the card already speaks for, so they render no second row. */ + consumedResults: Set +} + +const NO_EDIT_CARDS: EditCardModel = { editCards: new Map(), consumedResults: new Set() } + +/** An edit renders as one card, so its result block is folded into the call. The + * model decides which calls have landed; a call that has not keeps the generic + * tool view, its result still visible as the provider's own error. */ +function buildEditCards(blocks: NativeChatBlock[]): EditCardModel { + const editCards: EditCardModel['editCards'] = new Map() + const consumedResults: EditCardModel['consumedResults'] = new Set() + for (const [index, pair] of pairToolBlocks(blocks).entries()) { + const call = pair.call + if (!call || !isEditToolName(call.name)) { + continue + } + const files = editFilesFromToolPair({ + name: call.name, + input: call.input, + ...(call.state ? { state: call.state } : {}), + ...(pair.result + ? { + result: { + output: pair.result.output, + isError: pair.result.isError, + editPatch: pair.result.editPatch + } + } + : {}) + }) + if (!files || files.length === 0) { + continue + } + editCards.set(call, { files, key: `${call.name}:${index}` }) + if (pair.result) { + consumedResults.add(pair.result) + } + } + return { editCards, consumedResults } +} + /** A run of a message's tool calls/results, collapsed to a one-line summary that * expands to the individual inline tool lines. `expandSignal` lets the global * toolbar toggle drive every run at once while still allowing per-run override. */ @@ -160,6 +211,12 @@ export function NativeChatToolRun({ // The turn caret opens the activity group, while each child tool remains // collapsed. The global expand toolbar still opens child details together. const expandToolLines = expandOverride === undefined ? open : false + // Diffing every edit is the run's most expensive work, so a collapsed run — + // which renders none of it — never pays for it. + const { editCards, consumedResults } = useMemo( + () => (open ? buildEditCards(blocks) : NO_EDIT_CARDS), + [open, blocks] + ) const ActiveToolIcon = latestActiveCall && isCommandToolName(latestActiveCall.name) ? SquareTerminal : Wrench const fallbackLabel = @@ -233,6 +290,23 @@ export function NativeChatToolRun({ {(() => { const seen = new Map() return blocks.map((block) => { + const edit = editCards.get(block) + if (edit) { + return ( +
+ {edit.files.map((file, fileIndex) => ( + + ))} +
+ ) + } + if (consumedResults.has(block)) { + return null + } const signature = block.type === 'tool-call' ? `${block.type}:${block.name}:${JSON.stringify(block.input)}` diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 95de4fab458..243771fee8f 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16958,7 +16958,14 @@ "ranCommandsOneToolSummary": "Ran {{commandCount}} commands and used {{toolCount}} tool", "ranCommandsManyToolsSummary": "Ran {{commandCount}} commands and used {{toolCount}} tools", "usedOneSummary": "Used 1 tool", - "usedManySummary": "Used {{toolCount}} tools" + "usedManySummary": "Used {{toolCount}} tools", + "editedFile": "Edited file", + "addedFile": "Added file", + "deletedFile": "Deleted file", + "renamedFile": "Renamed file", + "copyDiff": "Copy diff", + "diffGap": "Lines not shown", + "diffTruncated": "Diff truncated" }, "providerFrame": { "byteLength": "{{value0}} bytes" diff --git a/src/shared/native-chat-begin-patch.ts b/src/shared/native-chat-begin-patch.ts new file mode 100644 index 00000000000..becd0ece46a --- /dev/null +++ b/src/shared/native-chat-begin-patch.ts @@ -0,0 +1,173 @@ +import { finalizeEditFile, type NativeChatEditFile } from './native-chat-edit-model' +import { editLinesFromUnifiedPatch, editLinesFromWholeFile } from './native-chat-unified-patch' + +const BEGIN = '*** Begin Patch' +const END = '*** End Patch' +const FILE_HEADER = /^\*\*\* (Add|Update|Delete) File: (.+)$/ +const MOVE_HEADER = /^\*\*\* Move to: (.+)$/ +/** Envelope structure that carries no file content of its own. */ +const CONTROL_LINE = /^\*\*\* (?:End of File|Environment ID:)/ + +/** The envelope reaches a command tool as one of its patch or command + * arguments, either whole or as one word of the argument vector it runs. + * Recover its text. Callers must gate this on the tool being one that runs a + * patch: a file's own contents may quote an envelope. + * + * `requireApplyCommand` is for a general command tool, where the envelope + * proves nothing on its own — a command writing documentation quotes one + * without applying it, and the command must say it is applying it. */ +export function unwrapBeginPatch( + input: unknown, + options?: { requireApplyCommand?: boolean } +): string | null { + const source = envelopeSource(input, options?.requireApplyCommand === true) + if (!source) { + return null + } + const start = source.indexOf(BEGIN) + if (start === -1) { + return null + } + const end = source.indexOf(END, start) + if (end === -1) { + // Without the closing marker there is nothing separating the patch body from + // whatever the command line continues with, and trailing shell syntax would + // render as file content the agent never wrote. + return null + } + return source.slice(start, end + END.length) +} + +/** The arguments that carry a patch or the command line that applies one. Only + * these are searched: any other value is data the tool operates on, and a file + * whose own contents quote an envelope would otherwise be read as a patch + * against some other file entirely. */ +const ENVELOPE_ARGUMENTS = ['input', 'command', 'patch', 'arguments', 'script'] as const +/** What a command tool runs to apply an envelope, as opposed to quoting one. + * Both spellings the runner accepts, since either one really applies it. */ +const APPLY_COMMAND = /apply_?patch/ + +/** The call payload may itself be a string holding JSON. Decoding it here, in + * the one consumer that needs its structure, keeps every other reader of the + * call input seeing exactly what the provider sent. */ +function envelopeSource(input: unknown, requireApplyCommand: boolean): string | null { + if (typeof input === 'string') { + const record = jsonRecord(input) + return record + ? envelopeArgument(record, requireApplyCommand) + : applied(input, requireApplyCommand) + } + return typeof input === 'object' && input !== null + ? envelopeArgument(input as Record, requireApplyCommand) + : null +} + +function applied(value: string, requireApplyCommand: boolean): string | null { + return !requireApplyCommand || APPLY_COMMAND.test(value) ? value : null +} + +function jsonRecord(value: string): Record | null { + if (!value.trimStart().startsWith('{')) { + return null + } + try { + const parsed: unknown = JSON.parse(value) + return typeof parsed === 'object' && parsed !== null && !Array.isArray(parsed) + ? (parsed as Record) + : null + } catch { + return null + } +} + +/** A command tool's argument is a vector, so the envelope sits one level in and + * the words that apply it may be a different element than the envelope. */ +function envelopeArgument( + record: Record, + requireApplyCommand: boolean +): string | null { + for (const key of ENVELOPE_ARGUMENTS) { + const value = record[key] + const words = + typeof value === 'string' + ? [value] + : Array.isArray(value) + ? value.filter((entry): entry is string => typeof entry === 'string') + : [] + if (requireApplyCommand && !words.some((word) => APPLY_COMMAND.test(word))) { + continue + } + const word = words.find((entry) => entry.includes(BEGIN)) + if (word) { + return word + } + } + return null +} + +/** Splits a `*** Begin Patch` envelope into one entry per file it touches. */ +export function editFilesFromBeginPatch(envelope: string): NativeChatEditFile[] { + const sections: { kind: 'Add' | 'Update' | 'Delete'; path: string; body: string[] }[] = [] + const moves = new Map() + + // Split on both newline forms once, so every marker below can be matched + // exactly rather than each pattern having to tolerate a trailing `\r`. + for (const raw of envelope.split(/\r?\n/)) { + const header = FILE_HEADER.exec(raw) + if (header) { + sections.push({ + kind: header[1] as 'Add' | 'Update' | 'Delete', + path: header[2]!.trim(), + body: [] + }) + continue + } + const move = MOVE_HEADER.exec(raw) + if (move && sections.length > 0) { + moves.set(sections.length - 1, move[1]!.trim()) + continue + } + if (raw === BEGIN || raw === END || CONTROL_LINE.test(raw) || sections.length === 0) { + continue + } + sections.at(-1)!.body.push(raw) + } + + return sections.flatMap((section, index) => { + const body = section.body.join('\n') + const moved = moves.get(index) ?? null + if (section.kind === 'Add' || section.kind === 'Delete') { + const sign = section.kind === 'Add' ? '+' : '-' + // Add/Delete bodies carry a sign per line but no hunk header. + const stripped = section.body + .map((line) => (line.startsWith(sign) ? line.slice(1) : line)) + .join('\n') + const whole = editLinesFromWholeFile(stripped, section.kind === 'Add' ? 'add' : 'del') + return [ + finalizeEditFile({ + path: section.path, + oldPath: null, + changeKind: section.kind === 'Add' ? 'added' : 'deleted', + lines: whole.lines, + lineNumbersKnown: true, + truncated: whole.truncated + }) + ] + } + // The first chunk of an update may carry no hunk header at all, and a + // section may carry no body either. The envelope named the file, so it is + // reported with whatever rows it has rather than dropped from a multi-file + // envelope with nothing to say it went missing. + const parsed = editLinesFromUnifiedPatch(body, { implicitFirstHunk: true }) + return [ + finalizeEditFile({ + path: moved ?? section.path, + oldPath: moved ? section.path : null, + changeKind: moved ? 'renamed' : 'edited', + lines: parsed?.lines ?? [], + lineNumbersKnown: parsed?.lineNumbersKnown ?? false, + truncated: parsed?.truncated ?? false + }) + ] + }) +} diff --git a/src/shared/native-chat-diff.ts b/src/shared/native-chat-diff.ts index 1df82db4800..a6dde625a14 100644 --- a/src/shared/native-chat-diff.ts +++ b/src/shared/native-chat-diff.ts @@ -14,8 +14,11 @@ const DIFF_TRUNCATED_LINE: NativeChatDiffLine = { } const HUNK_HEADER = /^@@ -\d+(?:,\d+)? \+\d+(?:,\d+)? @@/ -// Lines that open a new file section, so any hunk before them has ended. -const FILE_SECTION_START = +/** Lines that open a new file section, so any hunk before them has ended. + * `--- `/`+++ ` are deliberately absent: inside a hunk they are content — a + * removed `-- comment` is emitted as `--- comment` — so they go through + * `isFileHeaderPair` instead. */ +export const FILE_SECTION_START = /^(?:diff |index |old mode |new mode |new file mode |deleted file mode |similarity index |dissimilarity index |rename |copy |Binary files )/ // Markdown thematic break or YAML document separator, not a marker. const BARE_RULE = /^(?:-{3,}|\+{3,})$/ @@ -28,11 +31,17 @@ type DiffStructure = { } /** - * Locates the `--- ` / `+++ ` file headers. A bare `---`/`+++` prefix - * is not enough to spot one: a removed line whose content began with `--` - * (SQL/Lua `-- comment`, C `--i`) is emitted as `---`. Real headers - * always come as an adjacent pair and never appear inside a hunk. + * True when the row at `index` opens a `--- ` / `+++ ` file header. A + * bare `---`/`+++` prefix is not enough to spot one: a removed line whose + * content began with `--` (SQL/Lua `-- comment`, C `--i`) is emitted as + * `---`. Real headers always come as an adjacent pair and never appear + * inside a hunk, so callers must check this only outside one. */ +export function isFileHeaderPair(lines: readonly string[], index: number): boolean { + return (lines[index] ?? '').startsWith('--- ') && (lines[index + 1] ?? '').startsWith('+++ ') +} + +/** Locates the file headers and rules that are structure rather than content. */ function scanDiffStructure(lines: string[]): DiffStructure { const metaIndices = new Set() let isStructuredDiff = false @@ -57,7 +66,7 @@ function scanDiffStructure(lines: string[]): DiffStructure { metaIndices.add(index) continue } - if (line.startsWith('--- ') && (lines[index + 1] ?? '').startsWith('+++ ')) { + if (isFileHeaderPair(lines, index)) { metaIndices.add(index) metaIndices.add(index + 1) index += 1 diff --git a/src/shared/native-chat-edit-lcs.ts b/src/shared/native-chat-edit-lcs.ts new file mode 100644 index 00000000000..733d5f5c700 --- /dev/null +++ b/src/shared/native-chat-edit-lcs.ts @@ -0,0 +1,105 @@ +import { + MAX_EDIT_DIFF_CELLS, + splitEditContent, + type NativeChatEditLine +} from './native-chat-edit-model' + +/** Line diff between two contents, interleaved with context. Numbers are + * positions within the given contents, so they locate rows in the file only + * when the caller passed whole files. */ +export function editLinesFromContents( + originalContent: string, + modifiedContent: string +): { lines: NativeChatEditLine[]; truncated: boolean } { + const original = splitEditContent(originalContent) + const modified = splitEditContent(modifiedContent) + const lines = + original.lines.length * modified.lines.length <= MAX_EDIT_DIFF_CELLS + ? lcsLines(original.lines, modified.lines) + : prefixSuffixLines(original.lines, modified.lines) + return { lines, truncated: original.truncated || modified.truncated } +} + +function context(text: string, oldNo: number, newNo: number): NativeChatEditLine { + return { kind: 'context', text, oldLineNumber: oldNo, newLineNumber: newNo } +} + +function removal(text: string, oldNo: number): NativeChatEditLine { + return { kind: 'del', text, oldLineNumber: oldNo, newLineNumber: null } +} + +function addition(text: string, newNo: number): NativeChatEditLine { + return { kind: 'add', text, oldLineNumber: null, newLineNumber: newNo } +} + +function lcsLines(original: string[], modified: string[]): NativeChatEditLine[] { + const width = modified.length + 1 + const dp = new Uint32Array((original.length + 1) * width) + for (let i = original.length - 1; i >= 0; i -= 1) { + for (let j = modified.length - 1; j >= 0; j -= 1) { + dp[i * width + j] = + original[i] === modified[j] + ? dp[(i + 1) * width + j + 1]! + 1 + : Math.max(dp[(i + 1) * width + j]!, dp[i * width + j + 1]!) + } + } + + const lines: NativeChatEditLine[] = [] + let oldIndex = 0 + let newIndex = 0 + while (oldIndex < original.length && newIndex < modified.length) { + if (original[oldIndex] === modified[newIndex]) { + lines.push(context(original[oldIndex] ?? '', oldIndex + 1, newIndex + 1)) + oldIndex += 1 + newIndex += 1 + } else if (dp[(oldIndex + 1) * width + newIndex]! >= dp[oldIndex * width + newIndex + 1]!) { + lines.push(removal(original[oldIndex] ?? '', oldIndex + 1)) + oldIndex += 1 + } else { + lines.push(addition(modified[newIndex] ?? '', newIndex + 1)) + newIndex += 1 + } + } + for (; oldIndex < original.length; oldIndex += 1) { + lines.push(removal(original[oldIndex] ?? '', oldIndex + 1)) + } + for (; newIndex < modified.length; newIndex += 1) { + lines.push(addition(modified[newIndex] ?? '', newIndex + 1)) + } + return lines +} + +function prefixSuffixLines(original: string[], modified: string[]): NativeChatEditLine[] { + let prefix = 0 + while ( + prefix < original.length && + prefix < modified.length && + original[prefix] === modified[prefix] + ) { + prefix += 1 + } + let suffix = 0 + while ( + suffix + prefix < original.length && + suffix + prefix < modified.length && + original[original.length - suffix - 1] === modified[modified.length - suffix - 1] + ) { + suffix += 1 + } + + const lines: NativeChatEditLine[] = [] + for (let i = 0; i < prefix; i += 1) { + lines.push(context(original[i] ?? '', i + 1, i + 1)) + } + for (let i = prefix; i < original.length - suffix; i += 1) { + lines.push(removal(original[i] ?? '', i + 1)) + } + for (let i = prefix; i < modified.length - suffix; i += 1) { + lines.push(addition(modified[i] ?? '', i + 1)) + } + for (let i = original.length - suffix; i < original.length; i += 1) { + const newIndex = modified.length - suffix + (i - (original.length - suffix)) + lines.push(context(original[i] ?? '', i + 1, newIndex + 1)) + } + return lines +} diff --git a/src/shared/native-chat-edit-model.ts b/src/shared/native-chat-edit-model.ts new file mode 100644 index 00000000000..6a26f87c599 --- /dev/null +++ b/src/shared/native-chat-edit-model.ts @@ -0,0 +1,107 @@ +/** One rendered diff row. Numbers are per side: a removed row has no new-side + * number and an added row has no old-side number. `gap` marks the break + * between two regions of the file, which are otherwise concatenated and read + * as one continuous block even as the gutter jumps hundreds of lines. */ +export type NativeChatEditLineKind = 'context' | 'add' | 'del' | 'gap' + +export type NativeChatEditLine = { + kind: NativeChatEditLineKind + text: string + oldLineNumber: number | null + newLineNumber: number | null +} + +export type NativeChatEditChangeKind = 'added' | 'deleted' | 'edited' | 'renamed' + +export type NativeChatEditFile = { + path: string + /** Set only when the change moved the file. */ + oldPath: string | null + changeKind: NativeChatEditChangeKind + lines: NativeChatEditLine[] + added: number + removed: number + /** False when the numbers locate a row inside a snippet rather than the file, + * which is the case whenever the provider gave us no resolved hunk ranges. */ + lineNumbersKnown: boolean + truncated: boolean +} + +export const MAX_EDIT_LINES = 2_000 +export const MAX_EDIT_CHARS = 96_000 +/** The LCS table is quadratic; above this a linear prefix/suffix diff is used. */ +export const MAX_EDIT_DIFF_CELLS = 200_000 + +/** Rows of a source string, plus whether it was clipped before splitting. */ +export type EditContentLines = { lines: string[]; truncated: boolean } + +/** The one row splitter for every edit shape. Splits on both newline forms so a + * CRLF file never carries a trailing `\r` into a row, where it would render as + * a stray character, defeat the phantom-row guard, and reach the clipboard. */ +export function splitEditContent(content: string): EditContentLines { + if (content.length === 0) { + return { lines: [], truncated: false } + } + const truncated = content.length > MAX_EDIT_CHARS + const body = truncated ? content.slice(0, MAX_EDIT_CHARS) : content + const lines = body.split(/\r?\n/) + // Tested against the clipped body: on the un-clipped string this popped a + // real line whenever the slice fired. + if (body.endsWith('\n')) { + lines.pop() + } + return { lines, truncated } +} + +/** The break between two regions of a file. Carries no text and no position. */ +function editGapLine(): NativeChatEditLine { + return { kind: 'gap', text: '', oldLineNumber: null, newLineNumber: null } +} + +/** Appends a gap when rows already exist, so the break never opens a diff or + * doubles up behind an empty region. */ +export function pushEditGap(lines: NativeChatEditLine[]): void { + if (lines.length > 0 && lines.at(-1)?.kind !== 'gap') { + lines.push(editGapLine()) + } +} + +/** Unified line numbering: a removed row is located on the old side, everything + * else on the new side. One column, so a replaced line repeats its number. */ +export function unifiedLineNumber(line: NativeChatEditLine): number | null { + return line.kind === 'del' ? line.oldLineNumber : (line.newLineNumber ?? line.oldLineNumber) +} + +export function finalizeEditFile( + input: Omit & { + /** Set when the source text was clipped before it became rows. */ + truncated?: boolean + } +): NativeChatEditFile { + const overLineCap = input.lines.length > MAX_EDIT_LINES + const truncated = overLineCap || input.truncated === true + const capped = overLineCap ? input.lines.slice(0, MAX_EDIT_LINES) : input.lines + // A gap marks a break between regions, so one at the end marks nothing. The + // row cap can leave one behind even when the source did not. + let end = capped.length + while (end > 0 && capped[end - 1]?.kind === 'gap') { + end -= 1 + } + const trimmed = end === capped.length ? capped : capped.slice(0, end) + // Without resolved ranges the numbers locate a row inside a snippet; dropping + // them keeps a plausible-looking wrong position out of the gutter, the copy + // text, and the row keys. + const lines = input.lineNumbersKnown + ? trimmed + : trimmed.map((line) => ({ ...line, oldLineNumber: null, newLineNumber: null })) + let added = 0 + let removed = 0 + for (const line of lines) { + if (line.kind === 'add') { + added += 1 + } else if (line.kind === 'del') { + removed += 1 + } + } + return { ...input, lines, added, removed, truncated } +} diff --git a/src/shared/native-chat-edit-normalize.test.ts b/src/shared/native-chat-edit-normalize.test.ts new file mode 100644 index 00000000000..f49e23be296 --- /dev/null +++ b/src/shared/native-chat-edit-normalize.test.ts @@ -0,0 +1,609 @@ +import { describe, expect, it } from 'vitest' +import { editFilesFromToolPair, isEditToolName } from './native-chat-edit-normalize' +import { MAX_EDIT_CHARS, unifiedLineNumber } from './native-chat-edit-model' +import { editLinesFromUnifiedPatch } from './native-chat-unified-patch' +import { unwrapBeginPatch } from './native-chat-begin-patch' + +const gutter = (files: ReturnType): (number | null)[] => + (files ?? []).flatMap((file) => file.lines.map((line) => unifiedLineNumber(line))) + +/** A card takes evidence the edit landed, so these cases report the call as + * complete. Cases about the lifecycle itself pass their own state. */ +const settledFiles = ( + pair: Parameters[0] +): ReturnType => + editFilesFromToolPair({ state: 'completed', ...pair }) + +describe('editLinesFromUnifiedPatch', () => { + it('numbers deletes from the old side and adds from the new side', () => { + const parsed = editLinesFromUnifiedPatch('@@ -12,3 +12,3 @@\n ctx\n-was\n+now\n tail') + expect(parsed?.lineNumbersKnown).toBe(true) + expect(parsed?.lines.map((line) => [line.kind, unifiedLineNumber(line)])).toEqual([ + ['context', 12], + ['del', 13], + ['add', 13], + ['context', 14] + ]) + }) + + it('leaves rows unnumbered when the hunk header carries no ranges', () => { + const parsed = editLinesFromUnifiedPatch('@@\n ctx\n-was\n+now') + expect(parsed?.lineNumbersKnown).toBe(false) + expect(parsed?.lines.every((line) => unifiedLineNumber(line) === null)).toBe(true) + }) + + it('returns null for text with no hunk header', () => { + expect(editLinesFromUnifiedPatch('just prose\n- a bullet')).toBeNull() + }) + + it('keeps the hunk open across a mid-hunk no-newline marker', () => { + const parsed = editLinesFromUnifiedPatch( + '@@ -1,2 +1,2 @@\n keep\n-old\n\\ No newline at end of file\n+new\n\\ No newline at end of file' + ) + expect(parsed?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['context', 'keep'], + ['del', 'old'], + ['add', 'new'] + ]) + }) + + it('reads a removed line that starts with `--` as content, not a file header', () => { + const parsed = editLinesFromUnifiedPatch('@@ -1,4 +1,3 @@\n keep\n--- comment\n-gone\n tail') + expect(parsed?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['context', 'keep'], + ['del', '-- comment'], + ['del', 'gone'], + ['context', 'tail'] + ]) + }) + + it('skips a real file header pair, which only appears outside a hunk', () => { + const parsed = editLinesFromUnifiedPatch( + 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1,1 +1,1 @@\n-was\n+now' + ) + expect(parsed?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['del', 'was'], + ['add', 'now'] + ]) + }) + + it('splits CRLF rows without leaving a carriage return or a phantom row', () => { + const parsed = editLinesFromUnifiedPatch('@@ -1,2 +1,2 @@\r\n ctx\r\n-was\r\n+now\r\n') + expect(parsed?.lines.map((line) => line.text)).toEqual(['ctx', 'was', 'now']) + }) + + it('reports truncation when the patch text runs past the character cap', () => { + const body = `@@ -1,1 +1,1 @@\n${'+x\n'.repeat(MAX_EDIT_CHARS)}` + expect(editLinesFromUnifiedPatch(body)?.truncated).toBe(true) + }) + + it('marks the break between hunks, and only between them', () => { + const parsed = editLinesFromUnifiedPatch( + '@@ -40,2 +40,2 @@\n keep\n-was\n@@ -310,2 +310,2 @@\n+now\n tail' + ) + expect(parsed?.lines.map((line) => [line.kind, unifiedLineNumber(line)])).toEqual([ + ['context', 40], + ['del', 41], + ['gap', null], + ['add', 310], + ['context', 311] + ]) + }) + + it('reads a body that opens with no hunk header as an unlocatable hunk', () => { + const parsed = editLinesFromUnifiedPatch('-was\n+now', { implicitFirstHunk: true }) + expect(parsed?.lines.map((line) => line.kind)).toEqual(['del', 'add']) + expect(parsed?.lineNumbersKnown).toBe(false) + // Without the option the same body is not a patch at all. + expect(editLinesFromUnifiedPatch('-was\n+now')).toBeNull() + }) +}) + +describe('unwrapBeginPatch', () => { + it('recovers an envelope carried in one word of an argument vector', () => { + const envelope = '*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y\n*** End Patch' + expect( + unwrapBeginPatch({ + command: ['bash', '-lc', `apply_patch <<'EOF'\n${envelope}\nEOF`], + workdir: '/repo' + }) + ).toBe(envelope) + }) + + it('leaves an already-decoded envelope alone', () => { + const plain = '*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y\n*** End Patch' + expect(unwrapBeginPatch(plain)).toBe(plain) + }) + + it('ignores input with no envelope', () => { + expect(unwrapBeginPatch('ls -la')).toBeNull() + }) + + it('declines an envelope with no closing marker rather than swallowing the command line', () => { + const command = 'bash -c "*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y" && echo ok' + expect(unwrapBeginPatch(command)).toBeNull() + expect(settledFiles({ name: 'shell', input: command })).toBeNull() + }) +}) + +describe('editFilesFromToolPair', () => { + it('renders an apply_patch run through a command tool, which produced no diff', () => { + const files = settledFiles({ + name: 'exec', + input: { + command: [ + 'bash', + '-lc', + "apply_patch <<'EOF'\n*** Begin Patch\n*** Update File: src/a.ts\n@@\n ctx\n-was\n+now\n*** End Patch\nEOF" + ] + } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.path).toBe('src/a.ts') + expect(files?.[0]?.changeKind).toBe('edited') + expect(files?.[0]?.added).toBe(1) + expect(files?.[0]?.removed).toBe(1) + // Codex hunk headers are context anchors, so no row may claim a file position. + expect(files?.[0]?.lineNumbersKnown).toBe(false) + }) + + it('numbers an added file from 1', () => { + const files = settledFiles({ + name: 'exec', + input: + "apply_patch <<'EOF'\n*** Begin Patch\n*** Add File: new.ts\n+one\n+two\n*** End Patch\nEOF" + }) + expect(files?.[0]?.changeKind).toBe('added') + expect(files?.[0]?.lineNumbersKnown).toBe(true) + expect(gutter(files)).toEqual([1, 2]) + }) + + it('keeps a file whose update body carries no hunk header', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\n*** Update File: first.ts\n ctx\n-was\n+now\n*** Update File: second.ts\n@@ -1,1 +1,1 @@\n-a\n+b\n*** End Patch' + } + }) + expect(files?.map((file) => file.path)).toEqual(['first.ts', 'second.ts']) + expect(files?.[0]?.lines.map((line) => line.kind)).toEqual(['context', 'del', 'add']) + expect(files?.[0]?.lineNumbersKnown).toBe(false) + }) + + it('does not render envelope control lines as file content', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\n*** Environment ID: abc123\n*** Update File: a.ts\n@@\n-was\n+now\n*** End of File\n*** End Patch' + } + }) + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['was', 'now']) + }) + + it('reports a delete that names the file and carries no body', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { input: '*** Begin Patch\n*** Delete File: gone.ts\n*** End Patch' } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.changeKind).toBe('deleted') + expect(files?.[0]?.path).toBe('gone.ts') + expect(files?.[0]?.lines).toEqual([]) + }) + + it('reads a CRLF envelope, whose markers otherwise match nothing', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\r\n*** Update File: a.ts\r\n@@ -1,2 +1,2 @@\r\n-was\r\n+now\r\n*** End Patch' + } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.path).toBe('a.ts') + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['was', 'now']) + }) + + it('marks the break between resolved hunks that sit far apart', () => { + const files = settledFiles({ + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + result: { + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 42, oldLines: 1, newStart: 42, newLines: 1, lines: ['-was', '+now'] }, + { oldStart: 310, oldLines: 1, newStart: 310, newLines: 1, lines: ['-old', '+new'] } + ] + } + } + }) + expect(files?.[0]?.lines.map((line) => line.kind)).toEqual(['del', 'add', 'gap', 'del', 'add']) + // A break marks nothing at either end, and counts no change of its own. + expect(files?.[0]?.added).toBe(2) + expect(files?.[0]?.removed).toBe(2) + expect(gutter(files)).toEqual([42, 42, null, 310, 310]) + }) + + it('reads a move header as a rename', () => { + const files = settledFiles({ + name: 'exec', + input: + "apply_patch <<'EOF'\n*** Begin Patch\n*** Update File: old.ts\n*** Move to: new.ts\n@@\n-a\n+b\n*** End Patch\nEOF" + }) + expect(files?.[0]?.changeKind).toBe('renamed') + expect(files?.[0]?.oldPath).toBe('old.ts') + expect(files?.[0]?.path).toBe('new.ts') + }) + + it('interleaves a Claude snippet pair without claiming line positions', () => { + const files = settledFiles({ + name: 'Edit', + input: { + file_path: '/repo/a.ts', + old_string: 'keep\nwas\ntail', + new_string: 'keep\nnow\ntail' + } + }) + expect(files?.[0]?.lines.map((line) => line.kind)).toEqual(['context', 'del', 'add', 'context']) + expect(files?.[0]?.lineNumbersKnown).toBe(false) + }) + + it('prefers the resolved hunks on the result over the snippet pair', () => { + const files = settledFiles({ + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' }, + result: { + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { + oldStart: 12, + oldLines: 3, + newStart: 12, + newLines: 3, + lines: [' ctx', '-was', '+now', ' tail'] + } + ] + } + } + }) + expect(files?.[0]?.lineNumbersKnown).toBe(true) + expect(gutter(files)).toEqual([12, 13, 13, 14]) + }) + + it('treats a Write the provider reported as a creation as an added file', () => { + const files = settledFiles({ + name: 'Write', + input: { file_path: '/repo/new.ts', content: 'one\ntwo\n' }, + result: { output: 'File created successfully at: /repo/new.ts' } + }) + expect(files?.[0]?.changeKind).toBe('added') + expect(gutter(files)).toEqual([1, 2]) + }) + + it('does not claim a creation for a Write over an existing file', () => { + const overwrite = settledFiles({ + name: 'Write', + input: { file_path: '/repo/a.ts', content: 'one\ntwo\n' }, + result: { output: 'The file /repo/a.ts has been updated.' } + }) + expect(overwrite?.[0]?.changeKind).toBe('edited') + // With no result at all there is no evidence of a creation either. + const unreported = settledFiles({ + name: 'Write', + input: { file_path: '/repo/a.ts', content: 'one\n' } + }) + expect(unreported?.[0]?.changeKind).toBe('edited') + }) + + it('reads a MultiEdit, whose snippet pairs sit in edits[]', () => { + const files = settledFiles({ + name: 'MultiEdit', + input: { + file_path: '/repo/a.ts', + edits: [ + { old_string: 'was', new_string: 'now' }, + { old_string: 'gone', new_string: 'kept' } + ] + } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.path).toBe('/repo/a.ts') + // Each entry is its own region, so a break separates them. + expect(files?.[0]?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['del', 'was'], + ['add', 'now'], + ['gap', ''], + ['del', 'gone'], + ['add', 'kept'] + ]) + expect(files?.[0]?.added).toBe(2) + expect(files?.[0]?.removed).toBe(2) + }) + + it('leaves NotebookEdit to the generic tool view', () => { + expect(isEditToolName('NotebookEdit')).toBe(false) + }) + + it('drops the gutter numbers whenever they locate a snippet rather than the file', () => { + const files = settledFiles({ + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'keep\nwas', new_string: 'keep\nnow' } + }) + expect(files?.[0]?.lineNumbersKnown).toBe(false) + expect(gutter(files)).toEqual([null, null, null]) + expect( + files?.[0]?.lines.every((line) => line.oldLineNumber === null && line.newLineNumber === null) + ).toBe(true) + }) + + it('renders no card for an edit the provider rejected or has not landed', () => { + const failedInput = { file_path: '/repo/a.ts', old_string: 'missing', new_string: 'now' } + expect( + settledFiles({ + name: 'Edit', + input: failedInput, + result: { output: 'String to replace not found in file.', isError: true } + }) + ).toBeNull() + expect( + settledFiles({ + name: 'apply_patch', + input: { + changes: [{ path: 'a.ts', kind: { type: 'update' }, diff: '@@ -1 +1 @@\n-a\n+b' }] + }, + state: 'failed' + }) + ).toBeNull() + expect(settledFiles({ name: 'Edit', input: failedInput, state: 'running' })).toBeNull() + }) + + it('does not read a command tool result as a file edit', () => { + const patch = 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1 +1 @@\n-was\n+now' + expect( + settledFiles({ + name: 'exec', + input: { command: 'git diff' }, + result: { output: patch } + }) + ).toBeNull() + // The structured journal's `Diff` item carries its patch only on the result. + const diffed = settledFiles({ + name: 'Diff', + input: { path: '/repo/a.ts' }, + result: { output: patch } + }) + expect(diffed?.[0]?.path).toBe('/repo/a.ts') + expect(diffed?.[0]?.added).toBe(1) + }) + + it('reports truncation when the content runs past the character cap', () => { + const files = settledFiles({ + name: 'Write', + input: { file_path: '/repo/a.ts', content: `${'x'.repeat(MAX_EDIT_CHARS)}\nlast\n` } + }) + expect(files?.[0]?.truncated).toBe(true) + // The clipped body ends mid-line, so its one row is real and must survive. + expect(files?.[0]?.lines).toHaveLength(1) + }) + + it('reads Codex structured changes, stripping the move marker from the body', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + changes: [ + { + path: 'old.ts', + kind: { type: 'update', move_path: 'new.ts' }, + diff: '@@ -1,2 +1,2 @@\n-a\n+b\n\nMoved to: new.ts' + } + ] + } + }) + expect(files?.[0]?.changeKind).toBe('renamed') + expect(files?.[0]?.lines.some((line) => line.text.includes('Moved to'))).toBe(false) + expect(gutter(files)).toEqual([1, 1]) + }) + + it('reads a Codex add change, which arrives as raw content with no hunk header', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { changes: [{ path: 'new.ts', kind: { type: 'add' }, diff: 'one\ntwo' }] } + }) + expect(files?.[0]?.changeKind).toBe('added') + expect(files?.[0]?.added).toBe(2) + }) + + it('renders the file a write actually wrote, not one its content quotes', () => { + const files = settledFiles({ + name: 'Write', + input: { + file_path: 'docs/patch-format.md', + content: + 'Example:\n\n*** Begin Patch\n*** Update File: src/victim.ts\n@@\n-a\n+b\n*** End Patch\n' + }, + result: { output: 'File created successfully at: docs/patch-format.md' } + }) + expect(files?.map((file) => file.path)).toEqual(['docs/patch-format.md']) + expect(files?.[0]?.lines.some((line) => line.text.includes('Begin Patch'))).toBe(true) + }) + + it('finds an envelope in a command payload that arrived as JSON text', () => { + const envelope = '*** Begin Patch\n*** Update File: src/a.ts\n@@\n-was\n+now\n*** End Patch' + const files = settledFiles({ + name: 'shell', + input: JSON.stringify({ + command: ['bash', '-lc', `apply_patch <<'EOF'\n${envelope}\nEOF`], + workdir: '/repo' + }) + }) + expect(files?.[0]?.path).toBe('src/a.ts') + expect(files?.[0]?.added).toBe(1) + }) + + it('renders no card for a call the turn never answered', () => { + const input = { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' } + // No lifecycle and no result: nothing says the edit was applied. + expect(editFilesFromToolPair({ name: 'Edit', input })).toBeNull() + expect(editFilesFromToolPair({ name: 'Edit', input, state: 'completed' })).toHaveLength(1) + expect(editFilesFromToolPair({ name: 'Edit', input, result: { output: 'ok' } })).toHaveLength(1) + }) + + it('splits a multi-file patch into one card per file', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: '/repo/one.ts' }, + result: { + output: + 'diff --git a/one.ts b/one.ts\n--- a/one.ts\n+++ b/one.ts\n@@ -1,1 +1,1 @@\n-first\n+FIRST\n' + + 'diff --git a/two.ts b/two.ts\n--- a/two.ts\n+++ b/two.ts\n@@ -10,1 +10,1 @@\n-second\n+SECOND' + } + }) + expect(files?.map((file) => file.path)).toEqual(['one.ts', 'two.ts']) + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['first', 'FIRST']) + expect(gutter(files?.slice(1) ?? null)).toEqual([10, 10]) + }) + + it('keeps a file whose envelope section carries no body at all', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\n*** Update File: first.ts\n@@\n-a\n+b\n*** Update File: second.ts\n*** Update File: third.ts\n@@\n-c\n+d\n*** End Patch' + } + }) + expect(files?.map((file) => file.path)).toEqual(['first.ts', 'second.ts', 'third.ts']) + expect(files?.[1]?.lines).toEqual([]) + }) + + it('splits a multi-file patch written without per-file preamble lines', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: '/repo/one.ts' }, + result: { + output: + '--- a/one.ts\n+++ b/one.ts\n@@ -1,1 +1,1 @@\n-first\n+FIRST\n' + + '--- a/two.ts\n+++ b/two.ts\n@@ -10,1 +10,1 @@\n-second\n+SECOND' + } + }) + expect(files?.map((file) => file.path)).toEqual(['one.ts', 'two.ts']) + expect(gutter(files?.slice(1) ?? null)).toEqual([10, 10]) + }) + + it('refuses a card when the call names a file count instead of a file', () => { + expect( + settledFiles({ + name: 'Diff', + input: { path: '2 files' }, + result: { output: '@@\n-a\n+b\n@@\n-c\n+d' } + }) + ).toBeNull() + }) + + it('reports a clipped patch as truncated instead of rendering its marker', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: 'src/a.ts' }, + result: { output: '@@ -1,3 +1,3 @@\n ctx\n-was\n+now\n… (48210 bytes)' } + }) + expect(files?.[0]?.truncated).toBe(true) + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['ctx', 'was', 'now']) + }) + + it('reads a move appended to the patch body as a rename, as the other lane does', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: 'src/old.ts' }, + result: { output: '@@ -1,1 +1,1 @@\n-a\n+b\n\nMoved to: src/new.ts' } + }) + expect(files?.[0]?.changeKind).toBe('renamed') + expect(files?.[0]?.path).toBe('src/new.ts') + expect(files?.[0]?.oldPath).toBe('src/old.ts') + expect(files?.[0]?.lines.some((line) => line.text.includes('Moved to'))).toBe(false) + }) + + it('does not read a row that merely mentions a move as one', () => { + const body = '@@ -1,2 +1,2 @@\n ctx\n+See Moved to: docs/archive/index.md' + const fromPatch = settledFiles({ + name: 'Diff', + input: { path: 'docs/index.md' }, + result: { output: body } + }) + const fromChanges = settledFiles({ + name: 'apply_patch', + input: { changes: [{ path: 'docs/index.md', kind: { type: 'update' }, diff: body }] } + }) + for (const files of [fromPatch, fromChanges]) { + expect(files?.[0]?.changeKind).toBe('edited') + expect(files?.[0]?.oldPath).toBeNull() + expect(files?.[0]?.path).toBe('docs/index.md') + expect(files?.[0]?.lines.at(-1)?.text).toBe('See Moved to: docs/archive/index.md') + } + }) + + it('accepts either spelling of the command that applies an envelope', () => { + const envelope = '*** Begin Patch\n*** Update File: src/a.ts\n@@\n-was\n+now\n*** End Patch' + const files = settledFiles({ + name: 'shell', + input: { command: ['bash', '-lc', `applypatch <<'EOF'\n${envelope}\nEOF`] } + }) + expect(files?.[0]?.path).toBe('src/a.ts') + }) + + it('keeps the header destination for a rename the call names by its old path', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: 'old.txt' }, + result: { + output: + 'diff --git a/old.txt b/new.txt\n--- a/old.txt\n+++ b/new.txt\n@@ -1,1 +1,1 @@\n-a\n+b' + } + }) + expect(files?.[0]?.path).toBe('new.txt') + expect(files?.[0]?.oldPath).toBe('old.txt') + }) + + it('does not read a command that only quotes an envelope as an edit', () => { + const envelope = '*** Begin Patch\n*** Update File: src/real.ts\n@@\n-a\n+b\n*** End Patch' + expect( + settledFiles({ + name: 'shell', + input: { command: ['bash', '-lc', `cat > notes.md <<'EOF'\n${envelope}\nEOF`] } + }) + ).toBeNull() + }) + + it('does not call two compared directories a rename', () => { + const files = settledFiles({ + name: 'Diff', + input: {}, + result: { output: '--- d1/x.ts\n+++ d2/x.ts\n@@ -1,1 +1,1 @@\n-a\n+b' } + }) + expect(files?.[0]?.changeKind).toBe('edited') + expect(files?.[0]?.oldPath).toBeNull() + expect(files?.[0]?.path).toBe('d2/x.ts') + }) + + it('lets the call name the file when a preamble precedes the only header', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: '/repo/one.ts' }, + result: { + output: + 'warning: something\ndiff --git a/one.ts b/one.ts\n--- a/one.ts\n+++ b/one.ts\n@@ -1,1 +1,1 @@\n-a\n+b' + } + }) + // The preamble is its own nameless section, and must not make this look + // like a patch over several files. + expect(files?.map((file) => file.path)).toEqual(['/repo/one.ts']) + }) + + it('returns null for a tool that did not edit a file', () => { + expect(settledFiles({ name: 'Bash', input: { command: 'ls' } })).toBeNull() + expect(isEditToolName('Bash')).toBe(false) + expect(isEditToolName('Edit')).toBe(true) + }) +}) diff --git a/src/shared/native-chat-edit-normalize.ts b/src/shared/native-chat-edit-normalize.ts new file mode 100644 index 00000000000..95c2ca18bd5 --- /dev/null +++ b/src/shared/native-chat-edit-normalize.ts @@ -0,0 +1,349 @@ +import { editFilesFromBeginPatch, unwrapBeginPatch } from './native-chat-begin-patch' +import { editLinesFromContents } from './native-chat-edit-lcs' +import { + finalizeEditFile, + pushEditGap, + type NativeChatEditFile, + type NativeChatEditLine +} from './native-chat-edit-model' +import { stripBoundedTextMarker } from './structured-agent-session-projection' +import { + editLinesFromUnifiedPatch, + editLinesFromWholeFile, + unifiedPatchSections, + type UnifiedPatchSection +} from './native-chat-unified-patch' +import type { NativeChatEditPatch } from './native-chat-types' + +// `NotebookEdit` is deliberately absent: its input carries only the new cell +// source, so a card would render an unchanged cell as wholly added. It falls +// through to the generic tool view instead. +const CLAUDE_EDIT_TOOLS = new Set(['Edit', 'MultiEdit', 'Write', 'str_replace']) +/** Command tools, which run a patch as one of many things they can run, so a + * quoted envelope is not evidence that one was applied. */ +const COMMAND_PATCH_TOOLS = new Set(['exec', 'shell', 'local_shell']) +/** Tools whose input may wrap a `*** Begin Patch` envelope. The dedicated patch + * tool applies whatever it is given; a command tool must say that it is. */ +const PATCH_ENVELOPE_TOOLS = new Set(['apply_patch', ...COMMAND_PATCH_TOOLS]) +/** A count standing in for a path, from a producer that joined several files' + * patches and kept no per-file path. */ +const FILE_COUNT_PATH = /^\d+ files?$/ +/** Tools whose whole payload is patch text. `Diff` reaches its patch only + * through the result, because the structured journal projects a diff item as a + * call carrying just the path. */ +const PATCH_TEXT_TOOLS = new Set(['apply_patch', 'Diff']) + +export function isEditToolName(name: string): boolean { + return CLAUDE_EDIT_TOOLS.has(name) || PATCH_ENVELOPE_TOOLS.has(name) || PATCH_TEXT_TOOLS.has(name) +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null ? (value as Record) : null +} + +function text(value: unknown): string | null { + return typeof value === 'string' ? value : null +} + +/** Rows straight from resolved hunks, which is the only path with true numbers + * for a provider that reports its edits as a snippet pair. */ +function linesFromEditPatch(patch: NativeChatEditPatch): NativeChatEditLine[] { + const lines: NativeChatEditLine[] = [] + for (const hunk of patch.hunks) { + // Hunks are separate regions of the file; run together the gutter jumps + // from one to the next with nothing marking the skipped span. + pushEditGap(lines) + let oldNo = hunk.oldStart + let newNo = hunk.newStart + for (const raw of hunk.lines) { + if (raw.startsWith('+')) { + lines.push({ kind: 'add', text: raw.slice(1), oldLineNumber: null, newLineNumber: newNo }) + newNo += 1 + } else if (raw.startsWith('-')) { + lines.push({ kind: 'del', text: raw.slice(1), oldLineNumber: oldNo, newLineNumber: null }) + oldNo += 1 + } else { + lines.push({ + kind: 'context', + text: raw.startsWith(' ') ? raw.slice(1) : raw, + oldLineNumber: oldNo, + newLineNumber: newNo + }) + oldNo += 1 + newNo += 1 + } + } + } + return lines +} + +/** A whole-content write looks identical whether it created the file or + * overwrote one, so only positive evidence may claim a creation. With no + * evidence either way this errs toward the weaker claim: calling a creation an + * edit is imprecise, while calling an overwrite a creation is false and paints + * an existing file as wholly new. */ +const CREATED_FILE_RESULT = /^\s*File created successfully/ + +function wholeContentChangeKind( + input: Record, + output: string | undefined +): 'added' | 'edited' { + if (text(input.command) === 'create') { + return 'added' + } + return output !== undefined && CREATED_FILE_RESULT.test(output) ? 'added' : 'edited' +} + +/** `MultiEdit` carries its snippet pairs in `edits[]`, not at the top level. */ +function multiEditFiles(input: Record, path: string): NativeChatEditFile[] | null { + if (!Array.isArray(input.edits)) { + return null + } + const lines: NativeChatEditLine[] = [] + let truncated = false + for (const entry of input.edits) { + const edit = record(entry) + const oldString = text(edit?.old_string) ?? text(edit?.oldString) + const newString = text(edit?.new_string) ?? text(edit?.newString) + if (oldString === null && newString === null) { + continue + } + // Each entry is its own snippet, so it starts a new region. + pushEditGap(lines) + const diffed = editLinesFromContents(oldString ?? '', newString ?? '') + lines.push(...diffed.lines) + truncated ||= diffed.truncated + } + if (lines.length === 0) { + return null + } + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: 'edited', + lines, + // A snippet pair cannot say where in the file it sits. + lineNumbersKnown: false, + truncated + }) + ] +} + +function claudeEditFiles( + name: string, + input: Record, + output: string | undefined +): NativeChatEditFile[] | null { + const path = text(input.file_path) ?? text(input.path) ?? 'file' + if (name === 'MultiEdit') { + return multiEditFiles(input, path) + } + const oldString = text(input.old_string) ?? text(input.oldString) + const newString = text(input.new_string) ?? text(input.newString) + const content = text(input.content) ?? text(input.file_text) + if (oldString === null && content !== null) { + const whole = editLinesFromWholeFile(content, 'add') + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: wholeContentChangeKind(input, output), + lines: whole.lines, + lineNumbersKnown: true, + truncated: whole.truncated + }) + ] + } + if (oldString === null && newString === null) { + return null + } + const diffed = editLinesFromContents(oldString ?? '', newString ?? content ?? '') + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: 'edited', + lines: diffed.lines, + // A snippet pair cannot say where in the file it sits. + lineNumbersKnown: false, + truncated: diffed.truncated + }) + ] +} + +/** A move is appended to the patch body as prose rather than a header field, on + * every lane that carries the body as text. Left in place it renders as a + * numbered line of the file it moved. + * + * Anchored to the start of the final line: unanchored, a row whose own content + * mentions a move was cut in half and the file it names claimed as a rename + * that never happened. */ +const MOVE_MARKER = /(?:^|\n)Moved to: (.+)$/ + +function splitMoveMarker(patch: string): { body: string; movedTo: string | null } { + const match = MOVE_MARKER.exec(patch) + return match + ? { body: patch.slice(0, match.index), movedTo: match[1]!.trim() } + : { body: patch, movedTo: null } +} + +function codexChangeFiles(changes: unknown[]): NativeChatEditFile[] { + return changes.flatMap((entry) => { + const change = record(entry) + const path = text(change?.path) + const diff = text(change?.diff) + if (!change || !path || !diff) { + return [] + } + const kind = record(change.kind) + const kindType = text(kind?.type) ?? text(change.kind) ?? 'update' + const movePath = text(kind?.move_path) ?? text(change.movePath) + if (kindType === 'add' || kindType === 'delete') { + // Add and delete arrive as raw file content, with no hunk header or signs. + const whole = editLinesFromWholeFile(diff, kindType === 'add' ? 'add' : 'del') + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: kindType === 'add' ? 'added' : 'deleted', + lines: whole.lines, + lineNumbersKnown: true, + truncated: whole.truncated + }) + ] + } + const parsed = editLinesFromUnifiedPatch(splitMoveMarker(diff).body) + if (!parsed) { + return [] + } + return [ + finalizeEditFile({ + path: movePath ?? path, + oldPath: movePath ? path : null, + changeKind: movePath ? 'renamed' : 'edited', + lines: parsed.lines, + lineNumbersKnown: parsed.lineNumbersKnown, + truncated: parsed.truncated + }) + ] + }) +} + +/** One diff model for a tool call and its result, across every shape the + * supported agents use to report a file edit. */ +export function editFilesFromToolPair(pair: { + name: string + input: unknown + /** Provider lifecycle for the call, when the lane reports one. */ + state?: 'running' | 'completed' | 'failed' + result?: { output?: string; isError?: boolean; editPatch?: NativeChatEditPatch } +}): NativeChatEditFile[] | null { + // A card states the edit as made, so it takes evidence that it landed: the + // provider reporting the call complete, or a result that is not an error. + // Anything else — failed, still running, or a turn that stopped before the + // call was answered — keeps the generic tool view and its error body. + if (pair.state === 'failed' || pair.state === 'running' || pair.result?.isError === true) { + return null + } + if (pair.state !== 'completed' && pair.result === undefined) { + return null + } + const input = record(pair.input) + const patch = pair.result?.editPatch + if (patch && patch.hunks.length > 0) { + return [ + finalizeEditFile({ + path: patch.filePath ?? text(input?.file_path) ?? 'file', + oldPath: null, + changeKind: 'edited', + lines: linesFromEditPatch(patch), + lineNumbersKnown: true + }) + ] + } + + // Only a tool that runs a patch may be searched for an envelope: a file's own + // contents can quote one, and scanning a write's payload rendered a card for + // the quoted file while the file actually written never appeared. + if (PATCH_ENVELOPE_TOOLS.has(pair.name)) { + const envelope = unwrapBeginPatch(pair.input, { + requireApplyCommand: COMMAND_PATCH_TOOLS.has(pair.name) + }) + const files = envelope ? editFilesFromBeginPatch(envelope) : [] + if (files.length > 0) { + return files + } + } + + if (input && Array.isArray(input.changes)) { + const files = codexChangeFiles(input.changes) + if (files.length > 0) { + return files + } + } + + if (input && CLAUDE_EDIT_TOOLS.has(pair.name)) { + return claudeEditFiles(pair.name, input, pair.result?.output) + } + + if (!PATCH_TEXT_TOOLS.has(pair.name)) { + return null + } + // The result fallback is scoped to `Diff`, whose call carries only a path. + // Reading any command tool's output as a patch reclassified `git diff` as a + // file edit and swallowed the command line with it. + const patchText = + text(input?.patch) ?? text(input?.diff) ?? (pair.name === 'Diff' ? pair.result?.output : null) + if (!patchText) { + return null + } + // The body carries its own marker when the journal clipped it. Read as + // content it becomes a numbered line of the file, and the rows that follow + // are reported complete. + const bounded = stripBoundedTextMarker(patchText) + const moved = splitMoveMarker(bounded.text) + // One card per file the patch touches: run together, the later files' rows + // and gutter numbers sit under the first file's name. + const split = unifiedPatchSections(moved.body) + const callerPath = text(input?.path) ?? text(input?.file_path) + if (callerPath !== null && FILE_COUNT_PATH.test(callerPath)) { + // The producer joined several files' patches and kept a count in place of a + // path, so nothing here can name a file. Naming the card after the count + // would assert a file that does not exist. + return null + } + // A patch that names one file is the file the call is reporting on, so the + // call's own path wins — it is the provider's, where the header's is relative + // to the patch. A patch naming several has no one path, and a rename's + // destination is only ever in the header. Sections that name nothing are + // preamble and must not change that count. + const namedSections = split.sections.filter((section) => section.path !== null).length + const named = (section: UnifiedPatchSection): string => + (namedSections <= 1 && section.oldPath === null + ? (callerPath ?? section.path) + : (section.path ?? callerPath)) ?? 'file' + const files = split.sections.flatMap((section) => { + const parsed = editLinesFromUnifiedPatch(section.body) + if (!parsed && section.path === null) { + return [] + } + return [ + finalizeEditFile({ + path: named(section), + oldPath: section.oldPath, + changeKind: section.changeKind, + lines: parsed?.lines ?? [], + lineNumbersKnown: parsed?.lineNumbersKnown ?? false, + truncated: bounded.truncated || split.truncated || (parsed?.truncated ?? false) + }) + ] + }) + // The move marker names where the whole patch moved, so it can only speak for + // a patch describing one file. + if (moved.movedTo !== null && files.length === 1 && files[0]) { + const only = files[0] + return [{ ...only, path: moved.movedTo, oldPath: only.path, changeKind: 'renamed' }] + } + return files.length > 0 ? files : null +} diff --git a/src/shared/native-chat-types.ts b/src/shared/native-chat-types.ts index 5daa16f760e..124ee55dbe1 100644 --- a/src/shared/native-chat-types.ts +++ b/src/shared/native-chat-types.ts @@ -55,11 +55,31 @@ export type NativeChatToolCallBlock = { state?: 'running' | 'completed' | 'failed' } +/** One resolved hunk from a provider's edit result, carrying true file ranges. */ +export type NativeChatEditPatchHunk = { + oldStart: number + oldLines: number + newStart: number + newLines: number + /** Signed unified rows, as the provider emitted them. */ + lines: string[] +} + +/** Hunks the provider resolved against the real file before reporting the edit. + * Claude supplies these on its edit results; Codex resolves equivalently before + * sending, so its patch already carries ranges and needs no companion. */ +export type NativeChatEditPatch = { + filePath?: string + hunks: NativeChatEditPatchHunk[] +} + /** The result returned to the agent for a prior tool call. */ export type NativeChatToolResultBlock = { type: 'tool-result' output: string isError?: boolean + /** Present only for edit tools whose result reported resolved hunks. */ + editPatch?: NativeChatEditPatch } /** A reference to an image, by local path or remote URL. Exactly the field diff --git a/src/shared/native-chat-unified-patch.ts b/src/shared/native-chat-unified-patch.ts new file mode 100644 index 00000000000..2e34c79a6b3 --- /dev/null +++ b/src/shared/native-chat-unified-patch.ts @@ -0,0 +1,255 @@ +import { FILE_SECTION_START, isFileHeaderPair } from './native-chat-diff' +import { pushEditGap, splitEditContent, type NativeChatEditLine } from './native-chat-edit-model' + +const HUNK_RANGES = /^@@+ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@/ + +export type UnifiedPatchLines = { + lines: NativeChatEditLine[] + /** True only when every hunk carried real `@@` ranges. */ + lineNumbersKnown: boolean + /** The patch text was clipped before it became rows. */ + truncated: boolean +} + +/** Parses unified patch text, keeping the `@@` ranges as per-row line numbers. + * A hunk header whose `@@` is a bare context anchor with no ranges leaves its + * rows unnumbered rather than numbered from 1, because a wrong number reads as + * authoritative. + * + * `implicitFirstHunk` opens the body as a hunk of unknown position, for the + * patch dialect whose first chunk may carry no header at all. */ +export function editLinesFromUnifiedPatch( + text: string, + options?: { implicitFirstHunk?: boolean } +): UnifiedPatchLines | null { + const source = splitEditContent(text) + const rows = source.lines + const lines: NativeChatEditLine[] = [] + let oldNo: number | null = null + let newNo: number | null = null + let sawHunk = options?.implicitFirstHunk === true + let ranged = true + let inHunk = sawHunk + + for (let index = 0; index < rows.length; index += 1) { + const raw = rows[index] ?? '' + if (raw.startsWith('@@')) { + const match = HUNK_RANGES.exec(raw) + oldNo = match ? Number(match[1]) : null + newNo = match ? Number(match[3]) : null + // Successive hunks are separate regions of the file; concatenated with no + // break the gutter jumps and the reader sees one continuous block. + pushEditGap(lines) + sawHunk = true + inHunk = true + continue + } + // `\ No newline at end of file` sits mid-hunk, between the removed old last + // line and the added new one, so it ends nothing. + if (raw.startsWith('\\')) { + continue + } + if (!inHunk && isFileHeaderPair(rows, index)) { + index += 1 + continue + } + if (FILE_SECTION_START.test(raw)) { + inHunk = false + continue + } + if (!inHunk) { + continue + } + // Read off the rows rather than the header, so a body that opened with no + // header is reported as unlocatable just like a rangeless `@@`. + ranged &&= oldNo !== null || newNo !== null + if (raw.startsWith('+')) { + lines.push({ + kind: 'add', + text: raw.slice(1), + oldLineNumber: null, + newLineNumber: newNo + }) + newNo = newNo === null ? null : newNo + 1 + continue + } + if (raw.startsWith('-')) { + lines.push({ + kind: 'del', + text: raw.slice(1), + oldLineNumber: oldNo, + newLineNumber: null + }) + oldNo = oldNo === null ? null : oldNo + 1 + continue + } + lines.push({ + kind: 'context', + text: raw.startsWith(' ') ? raw.slice(1) : raw, + oldLineNumber: oldNo, + newLineNumber: newNo + }) + oldNo = oldNo === null ? null : oldNo + 1 + newNo = newNo === null ? null : newNo + 1 + } + + if (!sawHunk || lines.length === 0) { + return null + } + return { lines, lineNumbersKnown: ranged, truncated: source.truncated } +} + +const GIT_DIFF_HEADER = 'diff --git ' + +export type UnifiedPatchSection = { + /** Null when the patch text named no file, leaving it to the caller. */ + path: string | null + oldPath: string | null + changeKind: 'added' | 'deleted' | 'edited' | 'renamed' + body: string +} + +type Section = { + rows: string[] + oldPath: string | null + newPath: string | null + named: boolean + /** A `--- `/`+++ ` pair already named this section, so the next one is a new file. */ + hasHeaderPair: boolean + /** Only a `diff --git` header states both sides of a move as such. A bare + * pair with differing paths is just as likely two directories compared. */ + fromGitHeader: boolean +} + +/** Splits patch text into one section per file it touches. Without this a + * multi-file patch renders as a single card under the first file's name, with + * the later files' rows and gutter numbers beneath it. */ +export function unifiedPatchSections(text: string): { + sections: UnifiedPatchSection[] + truncated: boolean +} { + const source = splitEditContent(text) + const rows = source.lines + const sections: Section[] = [] + let current: Section | null = null + let inHunk = false + + const open = (): Section => { + const section: Section = { + rows: [], + oldPath: null, + newPath: null, + named: false, + hasHeaderPair: false, + fromGitHeader: false + } + sections.push(section) + return section + } + + for (let index = 0; index < rows.length; index += 1) { + const raw = rows[index] ?? '' + if (raw.startsWith(GIT_DIFF_HEADER)) { + const paths = gitHeaderPaths(raw) + current = open() + current.oldPath = paths.oldPath + current.newPath = paths.newPath + current.named = true + current.fromGitHeader = true + inHunk = false + continue + } + // A header pair is structure outside a hunk. Inside one it is also a file + // boundary, but only when a hunk header follows it immediately: a removed + // `-- x` over an added `++ y` is never followed by a column-0 `@@`, and + // that is what separates the files of a patch written without `diff --git` + // headers, where nothing else would end the previous file's hunk. + if (isFileHeaderPair(rows, index) && (!inHunk || (rows[index + 2] ?? '').startsWith('@@'))) { + // The pair names the section a `diff --git` just opened; a second pair in + // the same section is the next file of a patch written without them. + if (!current || current.hasHeaderPair) { + current = open() + } + current.oldPath = sourceHeaderPath(rows[index] ?? '') + current.newPath = sourceHeaderPath(rows[index + 1] ?? '') + current.named = true + current.hasHeaderPair = true + inHunk = false + index += 1 + continue + } + if (raw.startsWith('@@')) { + inHunk = true + } else if (FILE_SECTION_START.test(raw)) { + inHunk = false + } + current ??= open() + current.rows.push(raw) + } + + return { + sections: sections.map((section) => ({ + path: section.newPath ?? section.oldPath, + oldPath: sectionChangeKind(section) === 'renamed' ? section.oldPath : null, + changeKind: sectionChangeKind(section), + body: section.rows.join('\n') + })), + truncated: source.truncated + } +} + +function sectionChangeKind(section: Section): UnifiedPatchSection['changeKind'] { + if (!section.named) { + return 'edited' + } + if (section.newPath === null) { + return 'deleted' + } + if (section.oldPath === null) { + return 'added' + } + if (section.oldPath === section.newPath) { + return 'edited' + } + // Differing sides are a move only where the header says so. Bare pairs carry + // whatever paths the producer compared, which may be two directories. + return section.fromGitHeader ? 'renamed' : 'edited' +} + +/** `--- a/` / `+++ b/`, where the absent side is `/dev/null` and a + * trailing tab introduces the timestamp some producers append. */ +function sourceHeaderPath(line: string): string | null { + const value = (line.slice(4).split('\t')[0] ?? '').trim() + return value === '' || value === '/dev/null' ? null : value.replace(/^[ab]\//, '') +} + +function gitHeaderPaths(line: string): { oldPath: string | null; newPath: string | null } { + const rest = line.slice(GIT_DIFF_HEADER.length) + // Both halves carry the same path unless the file moved, so the second one + // starts at the last ` b/` rather than at the first space. + const split = rest.lastIndexOf(' b/') + if (split === -1) { + return { oldPath: null, newPath: null } + } + return { + oldPath: rest.slice(0, split).replace(/^a\//, ''), + newPath: rest.slice(split + 1).replace(/^b\//, '') + } +} + +/** Rows for a whole-file add or delete, which legitimately number from 1. */ +export function editLinesFromWholeFile( + content: string, + kind: 'add' | 'del' +): { lines: NativeChatEditLine[]; truncated: boolean } { + const body = splitEditContent(content) + return { + lines: body.lines.map((text, index) => ({ + kind, + text, + oldLineNumber: kind === 'del' ? index + 1 : null, + newLineNumber: kind === 'add' ? index + 1 : null + })), + truncated: body.truncated + } +} diff --git a/src/shared/structured-agent-session-projection.ts b/src/shared/structured-agent-session-projection.ts index 71cffa43762..94938dc955f 100644 --- a/src/shared/structured-agent-session-projection.ts +++ b/src/shared/structured-agent-session-projection.ts @@ -6,6 +6,22 @@ function boundedText(payload: { head: string; truncated: boolean; byteLength: nu return payload.truncated ? `${payload.head}\n… (${payload.byteLength} bytes)` : payload.head } +/** The markers a clipped payload carries in its own text, anchored to the end + * so nothing that merely looks like one inside the body can match. */ +const BOUNDED_TEXT_MARKERS = [ + /\n… \(\d+ bytes\)$/, + /\n\[Orca: output truncated — \d+ bytes total, digest [0-9a-f]+\]$/ +] + +/** Recovers the clipped body from a bounded payload's text, and says whether a + * marker was there. A reader that treats the text as content renders the + * marker as a line of it — with a line number, which reads as a real position + * in the file — and reports the body as complete. */ +export function stripBoundedTextMarker(text: string): { text: string; truncated: boolean } { + const stripped = BOUNDED_TEXT_MARKERS.reduce((value, marker) => value.replace(marker, ''), text) + return { text: stripped, truncated: stripped.length !== text.length } +} + function itemBlocks(item: AgentJournalRenderItem): { role: NativeChatMessage['role'] blocks: NativeChatBlock[] From 58553bfe1c0476b583ab6c17551c97603c62da3f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:45:42 -0700 Subject: [PATCH 13/26] fix(recovery): fail a renderer recovery reload that never loads, instead of leaving a dead window (#18466) --- .../expected-teardown-state.ts | 9 + src/main/startup/main-window-actions.ts | 18 +- src/main/startup/main-window-controller.ts | 31 +- ...reateMainWindow-close-confirmation.test.ts | 36 +- ...teMainWindow-markdown-editor-focus.test.ts | 32 +- ...ainWindow-recovery-reload-watchdog.test.ts | 671 ++++++++++++++++++ ...MainWindow-renderer-crash-recovery.test.ts | 42 +- .../createMainWindow-startup-reveal.test.ts | 4 +- ...eateMainWindow-system-resume-relay.test.ts | 4 +- ...ainWindow-terminal-focus-shortcuts.test.ts | 44 +- ...eateMainWindow-tray-minimize-close.test.ts | 4 +- ...ndow-zoom-and-tab-switch-shortcuts.test.ts | 36 +- src/main/window/createMainWindow.test.ts | 40 +- src/main/window/createMainWindow.ts | 35 +- src/main/window/main-window-contracts.ts | 38 +- .../window/main-window-focus-lifecycle.ts | 37 +- .../window/main-window-load-error-code.ts | 12 + .../window/renderer-recovery-prompt.test.ts | 52 +- src/main/window/renderer-recovery-prompt.ts | 68 +- .../renderer-recovery-reload-watchdog.test.ts | 136 ++++ .../renderer-recovery-reload-watchdog.ts | 310 ++++++++ src/renderer/src/i18n/locales/en.json | 12 + 22 files changed, 1495 insertions(+), 176 deletions(-) create mode 100644 src/main/window/createMainWindow-recovery-reload-watchdog.test.ts create mode 100644 src/main/window/main-window-load-error-code.ts create mode 100644 src/main/window/renderer-recovery-reload-watchdog.test.ts create mode 100644 src/main/window/renderer-recovery-reload-watchdog.ts diff --git a/src/main/crash-reporting/expected-teardown-state.ts b/src/main/crash-reporting/expected-teardown-state.ts index 1480ecbbfe2..c2993769633 100644 --- a/src/main/crash-reporting/expected-teardown-state.ts +++ b/src/main/crash-reporting/expected-teardown-state.ts @@ -10,9 +10,17 @@ type Clock = () => number const monotonicNow = (): number => performance.now() let now: Clock = monotonicNow let systemSessionEndedAt: number | null = null +let systemSessionEnded = false export function markSystemSessionEnding(): void { systemSessionEndedAt = now() + systemSessionEnded = true +} + +// Why latched, unlike the 5s crash-suppression window below: a native dialog or a recovery verdict is never +// right once the OS is tearing the session down, however long the process outlives the signal. +export function isSystemSessionEnding(): boolean { + return systemSessionEnded } function isRecentSystemSessionEnd(): boolean { @@ -55,4 +63,5 @@ export function resolveExpectedTeardownScope({ export function resetExpectedTeardownStateForTest(clock: Clock = monotonicNow): void { now = clock systemSessionEndedAt = null + systemSessionEnded = false } diff --git a/src/main/startup/main-window-actions.ts b/src/main/startup/main-window-actions.ts index 585fa0a754e..0acc8d1a074 100644 --- a/src/main/startup/main-window-actions.ts +++ b/src/main/startup/main-window-actions.ts @@ -16,7 +16,10 @@ import { describeInstallDirAclPoison, isBlockingInstallDirAclRepairInFlight } from './windows-install-dir-acl-recovery' -import { presentRendererRecoveryPrompt } from '../window/renderer-recovery-prompt' +import { + presentRendererRecoveryPrompt, + type RendererRecoveryPromptFailure +} from '../window/renderer-recovery-prompt' // The window module injects this callback to avoid a cycle between actions and lifecycle code. let openWindow: (options?: { revealOnDidFinishLoad?: boolean }) => BrowserWindow @@ -147,9 +150,14 @@ export function sendOpenCrashReport(targetWindow?: BrowserWindow | null): void { } // Why: on renderer crash-loop the breaker stops auto-reloading and the window goes blank, so a main-process dialog is the only retry/quit surface. -export async function showRendererRecoveryPrompt(recentRecoveryCount: number): Promise { +export async function showRendererRecoveryPrompt( + recentRecoveryCount: number, + failure?: RendererRecoveryPromptFailure, + retry?: () => void +): Promise { await presentRendererRecoveryPrompt({ recentRecoveryCount, + ...(failure ? { failure } : {}), isQuitting: () => state.isQuitting, diagnose: describeInstallDirAclPoison, showMessageBox: (options) => { @@ -164,6 +172,12 @@ export async function showRendererRecoveryPrompt(recentRecoveryCount: number): P } recordDurableCrashBreadcrumb('renderer_recovery_manual_retry') // Why: leave the breaker open so a re-crash re-raises this prompt instead of resuming the auto-reload loop. + // Why watched: Reload is the dialog's default button, and an unwatched retry that stalls returns the user to + // the same silent hang with no further prompt — the watchdog re-raises this dialog instead. + if (retry) { + retry() + return + } loadMainWindow(state.mainWindow) }, quit: () => { diff --git a/src/main/startup/main-window-controller.ts b/src/main/startup/main-window-controller.ts index 63d84df763c..8ea245fc256 100644 --- a/src/main/startup/main-window-controller.ts +++ b/src/main/startup/main-window-controller.ts @@ -113,13 +113,19 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} reason: details.reason, expectedTeardown: getExpectedTeardownScope(webContentsId, false) }), - onRendererRecoveryExhausted: ({ details, recentRecoveryCount }) => { - recordDurableCrashBreadcrumb('renderer_recovery_circuit_breaker_open', { - reason: details.reason, - exitCode: details.exitCode ?? null, - recentRecoveryCount - }) - void showRendererRecoveryPrompt(recentRecoveryCount) + onRendererRecoveryExhausted: ({ details, recentRecoveryCount, cause, retry }) => { + // Why two names: a stalled reload never opened the breaker, and a bundle that says it did misreads the failure. + recordDurableCrashBreadcrumb( + cause === 'reload-stalled' + ? 'renderer_recovery_reload_exhausted' + : 'renderer_recovery_circuit_breaker_open', + { + reason: details.reason, + exitCode: details.exitCode ?? null, + recentRecoveryCount + } + ) + void showRendererRecoveryPrompt(recentRecoveryCount, cause, retry) }, deferLoad: true, ...(options.revealOnDidFinishLoad === true ? { revealOnDidFinishLoad: true } : {}), @@ -131,9 +137,16 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} } recordCrashBreadcrumb('manual_reload_requested', { ignoreCache }) }, - onBeforeRecoveryReload: (webContentsId) => { + // Manual retries also preserve PTYs, but have their own intent breadcrumb. + onBeforeRecoveryReload: (webContentsId, trigger) => { markRecoveryReloadInFlight(webContentsId) - recordDurableCrashBreadcrumb('renderer_recovery_reload') + if (trigger === 'automatic') { + recordDurableCrashBreadcrumb('renderer_recovery_reload') + } + }, + // Pair the intent breadcrumb with its path-free outcome. + onRecoveryReloadOutcome: ({ status, ...outcome }) => { + recordDurableCrashBreadcrumb(`renderer_recovery_reload_${status}`, outcome) } }) recordCrashBreadcrumb('main_window_created') diff --git a/src/main/window/createMainWindow-close-confirmation.test.ts b/src/main/window/createMainWindow-close-confirmation.test.ts index a8c9f70c503..6f91dd4add8 100644 --- a/src/main/window/createMainWindow-close-confirmation.test.ts +++ b/src/main/window/createMainWindow-close-confirmation.test.ts @@ -73,8 +73,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const onQuitAborted = vi.fn() browserWindowMock.mockImplementation(function () { @@ -122,8 +122,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -178,8 +178,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn(), + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()), close: vi.fn(() => { windowHandlers.close({} as never) }) @@ -238,8 +238,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const updateUI = vi.fn() const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) @@ -296,8 +296,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -353,8 +353,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -401,8 +401,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -451,8 +451,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), destroy, - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) createMainWindow(null, { getIsQuitting: () => true }) @@ -500,8 +500,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), destroy, - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) createMainWindow(null, { getIsQuitting: () => true }) diff --git a/src/main/window/createMainWindow-markdown-editor-focus.test.ts b/src/main/window/createMainWindow-markdown-editor-focus.test.ts index 3189019f837..ef6fca229d3 100644 --- a/src/main/window/createMainWindow-markdown-editor-focus.test.ts +++ b/src/main/window/createMainWindow-markdown-editor-focus.test.ts @@ -54,8 +54,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -104,8 +104,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -157,8 +157,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -216,8 +216,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -275,8 +275,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -355,8 +355,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -432,8 +432,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -496,8 +496,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts b/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts new file mode 100644 index 00000000000..e551579a7ab --- /dev/null +++ b/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts @@ -0,0 +1,671 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as DurableCrashBreadcrumbModule from '../crash-reporting/durable-crash-breadcrumb' + +const { recordDurableCrashBreadcrumbMock } = vi.hoisted(() => ({ + recordDurableCrashBreadcrumbMock: vi.fn() +})) +vi.mock('../crash-reporting/durable-crash-breadcrumb', async (importOriginal) => ({ + ...(await importOriginal()), + recordDurableCrashBreadcrumb: recordDurableCrashBreadcrumbMock +})) + +vi.mock('electron', async () => + (await import('./createMainWindow-test-harness')).electronModuleMock() +) +vi.mock('@electron-toolkit/utils', async () => + (await import('./createMainWindow-test-harness')).electronToolkitUtilsMock() +) +vi.mock('./macos-tahoe-release', async () => + (await import('./createMainWindow-test-harness')).macosTahoeReleaseMock() +) +vi.mock('../app-icon', async () => (await import('./createMainWindow-test-harness')).appIconMock()) +vi.mock('../browser/browser-manager', async () => + (await import('./createMainWindow-test-harness')).browserManagerMock() +) +vi.mock('../browser/browser-client-page-renderer-runtime', async () => { + const harness = await import('./createMainWindow-test-harness') + return { + attachBrowserClientPageRenderer: harness.attachClientPageRendererMock, + retireBrowserClientPageRenderer: harness.retireClientPageRendererMock + } +}) + +import { createMainWindow } from './createMainWindow' +import { + browserWindowMock, + isMock, + powerMonitorOnMock, + resetMainWindowMocks +} from './createMainWindow-test-harness' +import { + RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS, + RENDERER_RECOVERY_LOAD_TIMEOUT_MS +} from './renderer-recovery-reload-watchdog' + +const DOCUMENT_URL = 'file:///opt/orca/renderer/index.html' +// A real macOS install URL: the crash-report redactor's PATH_PATTERNS provably leave this one intact. +const INSTALL_PATH_LOAD_ERROR = + "ERR_FILE_NOT_FOUND (-6) loading 'file:///Users/jane.doe/Applications/Orca.app/Contents/Resources/app.asar/out/renderer/index.html'" +const CRASH = { reason: 'crashed', exitCode: 5 } as Electron.RenderProcessGoneDetails + +/** + * Regression cover for the field failure: the recovery reload is issued, never produces a document, and nothing + * notices — no did-fail-load, no breaker (it counts renderer deaths only), no retry, no prompt. + */ +describe('renderer recovery reload watchdog', () => { + beforeEach(() => { + resetMainWindowMocks() + recordDurableCrashBreadcrumbMock.mockClear() + vi.useFakeTimers() + }) + + const createHarness = () => { + // Why fan-out: dom-ready and did-finish-load have several real registrants on this one webContents, so + // last-writer-wins would silently drop the watchdog's listener if registration order ever changed. + const registered: Record void)[]> = {} + const windowHandlers: Record void> = {} + const register = (event: string, handler: (...args: any[]) => void): void => { + const handlers = (registered[event] ??= []) + handlers.push(handler) + windowHandlers[event] ??= (...args: any[]) => { + for (const listener of handlers.slice()) { + listener(...args) + } + } + } + // Loads stay pending unless a test settles one: that is exactly the stall being reproduced. + const settleLoad: { resolve: () => void; reject: (error: Error) => void }[] = [] + const pendingLoad = (): Promise => + new Promise((resolve, reject) => settleLoad.push({ resolve, reject })) + const webContents = { + id: 143, + getURL: vi.fn(() => DOCUMENT_URL), + isDestroyed: vi.fn(() => false), + on: vi.fn(register), + setZoomLevel: vi.fn(), + setBackgroundThrottling: vi.fn(), + invalidate: vi.fn(), + setWindowOpenHandler: vi.fn(), + send: vi.fn() + } + const browserWindowInstance = { + webContents, + on: vi.fn(register), + isDestroyed: vi.fn(() => false), + isMaximized: vi.fn(() => true), + isFullScreen: vi.fn(() => false), + getSize: vi.fn(() => [1200, 800]), + setSize: vi.fn(), + maximize: vi.fn(), + show: vi.fn(), + setWindowButtonPosition: vi.fn(), + loadFile: vi.fn(pendingLoad), + loadURL: vi.fn(pendingLoad) + } + browserWindowMock.mockImplementation(function () { + return browserWindowInstance + }) + const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) + const crashRenderer = (): void => { + windowHandlers['render-process-gone']?.({} as never, CRASH) + vi.advanceTimersByTime(250) + } + const reachMilestone = (milestone: 'committed' | 'dom-ready'): void => + windowHandlers[milestone === 'committed' ? 'did-navigate' : 'dom-ready']?.() + return { + browserWindowInstance, + consoleError, + crashRenderer, + reachMilestone, + settleLoad, + windowHandlers + } + } + + it('retries once when the recovery reload never produces a document, then hands the user the prompt', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + // 1 initial load + 1 recovery reload, which now stalls forever. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(1) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS, + progress: 'none' + }) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 2 }) + ) + // Retry budget spent: stop reloading and surface the only retry/quit surface the user has. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith({ + details: CRASH, + webContentsId: 143, + recentRecoveryCount: 1, + cause: 'reload-stalled', + retry: expect.any(Function) + }) + + consoleError.mockRestore() + }) + + it('clears the watchdog when the recovery reload finishes loading', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(2_000) + settleLoad[1]?.resolve() + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'loaded', + attempt: 1, + elapsedMs: 2_000 + }) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 3) + expect(onRecoveryReloadOutcome).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + consoleError.mockRestore() + }) + + it('keeps watching the retry when a stale did-finish-load arrives after it was issued', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // did-finish-load carries no attempt token: this one belongs to the load the timer just abandoned. Crediting + // the retry with it disarms the watchdog over a load still in flight — the exact hole this watchdog closes. + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 2 }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('does not take an error page as the retry landing', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_FILE_NOT_FOUND (-6)')) + await vi.advanceTimersByTimeAsync(0) + // Chromium commits an error document for the failed load, and that document emits did-finish-load too. + windowHandlers['did-finish-load']?.() + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('raises one prompt, however many times recovery gives up underneath it', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // The renderer dies again while the box is up; the breaker never counted stalls, so it lets the reload go. + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + + // Nothing dismisses a native message box: a retry the user never asked for, or a second box, stacks on it. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + // The stall is still on the record, so the bundle does not read as a recovery that quietly worked. + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + + // Answering the box with Reload hands the next verdict back to the user. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + + it('still reloads from a crash-loop prompt raised after an earlier recovery had landed', async () => { + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + // Every recovery reload lands, and every landed document then dies with its renderer. + for (let attempt = 1; attempt <= 3; attempt += 1) { + crashRenderer() + settleLoad[attempt]?.resolve() + await vi.advanceTimersByTimeAsync(0) + } + crashRenderer() + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + const loads = browserWindowInstance.loadFile.mock.calls.length + + // The last document landed, but the renderer took it down: declining Reload here strands the user. + + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(loads + 1) + + consoleError.mockRestore() + }) + + it('does not stack a crash-loop prompt on one that is already up', () => { + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + for (let attempt = 0; attempt < 5; attempt += 1) { + crashRenderer() + } + + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('escalates a rejected recovery load immediately instead of waiting out the watchdog', async () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + settleLoad[1]?.reject(new Error("ERR_FILE_NOT_FOUND (-6) loading 'file:///opt/orca'")) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'failed', + attempt: 1, + errorCode: 'ERR_FILE_NOT_FOUND' + }) + ) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('ignores a superseded load rejection so ERR_ABORTED never escalates', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + // A second renderer death supersedes the first reload; Chromium rejects the abandoned load with ERR_ABORTED. + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + settleLoad[1]?.reject(new Error('ERR_ABORTED (-3)')) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('does not escalate when another navigation aborts the live recovery load', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad, windowHandlers } = + createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + // Chromium aborts the recovery load because something else replaced it — a user navigation, a close race, + // another loadURL caller. The attempt token still says this reload is live, so nothing else filters it. + settleLoad[1]?.reject(new Error(`ERR_ABORTED (-3) loading '${DOCUMENT_URL}'`)) + await vi.advanceTimersByTimeAsync(0) + + // A cold retry here would stomp the load that superseded this one. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + // The replacement load lands, and the window the user sees was never worth a Reload/Quit prompt. The crumb + // says so: elapsedMs measures the replacement, and the budget analysis has to be able to leave it out. + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', attempt: 1, superseded: true }) + ) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('still escalates on silence when an aborted recovery load has nothing behind it', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_ABORTED (-3)')) + await vi.advanceTimersByTimeAsync(0) + + // Ignoring the abort must not disarm the watchdog: the cap still bounds a load that goes nowhere. + await vi.advanceTimersByTimeAsync(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'reload-stalled' }) + ) + + consoleError.mockRestore() + }) + + it('gives the dev server a longer budget than a packaged load', () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + isMock.dev = true + vi.stubEnv('ELECTRON_RENDERER_URL', 'http://localhost:5173/') + + try { + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + expect(browserWindowInstance.loadURL).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + vi.advanceTimersByTime(1) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + } finally { + vi.unstubAllEnvs() + consoleError.mockRestore() + } + }) + + it('stays silent when the stalled window is already closing', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + windowHandlers.close?.({ preventDefault: vi.fn() } as never) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + it('keeps the install path out of the outcome breadcrumb', async () => { + const onRecoveryReloadOutcome = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + settleLoad[1]?.reject(new Error(INSTALL_PATH_LOAD_ERROR)) + await vi.advanceTimersByTimeAsync(0) + + const outcome = onRecoveryReloadOutcome.mock.calls[0]?.[0] + expect(outcome).toEqual({ + status: 'failed', + attempt: 1, + elapsedMs: 0, + progress: 'none', + errorCode: 'ERR_FILE_NOT_FOUND' + }) + // sanitizeCrashReportString cannot redact a file:///Users/... URL, so nothing path-shaped may reach the crumb. + expect(JSON.stringify(outcome)).not.toContain('/') + + consoleError.mockRestore() + }) + + it('records a durable breadcrumb for a rejected load, since console output never reaches the bundle', async () => { + const { consoleError, settleLoad } = createHarness() + + createMainWindow(null, {}) + settleLoad[0]?.reject(new Error(INSTALL_PATH_LOAD_ERROR)) + await vi.advanceTimersByTimeAsync(0) + + // Catching the rejection retired the main_unhandled_rejection crumb this used to produce. + expect(recordDurableCrashBreadcrumbMock).toHaveBeenCalledWith('main_window_load_failed', { + errorCode: 'ERR_FILE_NOT_FOUND' + }) + + consoleError.mockRestore() + }) + + it('escalates to the prompt when both attempts are rejected outright', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_CONNECTION_REFUSED (-102)')) + await vi.advanceTimersByTimeAsync(0) + settleLoad[2]?.reject(new Error('ERR_CONNECTION_REFUSED (-102)')) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'failed', attempt: 2, errorCode: 'ERR_CONNECTION_REFUSED' }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'reload-stalled', recentRecoveryCount: 1 }) + ) + + consoleError.mockRestore() + }) + + it('hands the prompt a watched retry so a stalled manual reload re-raises it', () => { + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // Reload is the dialog's default button; unwatched it returned the user to the same unbounded silent hang. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(5) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + + it('names the crash-loop cause and gives that prompt a watched retry too', () => { + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + for (let attempt = 0; attempt < 4; attempt += 1) { + crashRenderer() + } + + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'crash-loop' }) + ) + expect(typeof onRendererRecoveryExhausted.mock.calls[0]?.[0].retry).toBe('function') + + consoleError.mockRestore() + }) + + it('restarts the stall budget when the machine resumes mid-load', () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + // Sleep freezes the timer; on wake it would otherwise fire against a load that never got its budget. + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + const resume = powerMonitorOnMock.mock.calls.find(([event]) => event === 'resume')?.[1] as ( + ...args: unknown[] + ) => void + resume() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(1) + // Why the full span: rewriting the issue time on resume publishes time-since-wake into the bundle, which is + // silently wrong on any laptop — the outcome crumb exists to be honest about how long the load actually ran. + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2 - 1 + }) + ) + + consoleError.mockRestore() + }) + + it('never restarts a load that reached a document, and gives it the rest of the cap', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, reachMilestone } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(10_000) + reachMilestone('committed') + + // 'no did-finish-load yet' is not a stall: a cold restart here throws away a load that already committed, and + // a machine that would have landed at ~60s misses the budget entirely. + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + + reachMilestone('dom-ready') + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2, + progress: 'dom-ready' + }) + // Still never restarted, and the cap keeps the ~90s worst case the no-document path already had. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('records a reload that lands after the prompt, and leaves the recovered window alone', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + onRecoveryReloadOutcome.mockClear() + vi.advanceTimersByTime(30_000) + settleLoad[2]?.resolve() + await vi.advanceTimersByTimeAsync(0) + + // Nothing cancels a pending Chromium load, so escalation must keep watching: a bundle that reads + // `exhausted` for a recovery that actually worked misleads the next triage round. + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'loaded', + attempt: 2, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS + 30_000, + afterPrompt: true + }) + + // No API dismisses a native message box, so Reload is still aimed at a window that came back; taking it + // would destroy the session the recovery just restored. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('separates the automatic recovery reload from the prompt-driven retry', () => { + const onBeforeRecoveryReload = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onBeforeRecoveryReload, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + + // The field counts keyed on renderer_recovery_reload mean 'automatic recovery'; a manual retry recorded + // under the same name silently redefines them. + expect(onBeforeRecoveryReload.mock.calls.map(([, trigger]) => trigger)).toEqual([ + 'automatic', + 'automatic', + 'manual-retry' + ]) + + consoleError.mockRestore() + }) + + it('keeps a shutdown-aborted load out of the crash breadcrumb stream', async () => { + const { consoleError, settleLoad } = createHarness() + + createMainWindow(null, {}) + settleLoad[0]?.reject(new Error(`ERR_ABORTED (-3) loading '${DOCUMENT_URL}'`)) + await vi.advanceTimersByTimeAsync(0) + + // A quit or close aborts the in-flight startup load; a healthy shutdown must not look like a launch failure. + expect(recordDurableCrashBreadcrumbMock).not.toHaveBeenCalledWith( + 'main_window_load_failed', + expect.anything() + ) + + consoleError.mockRestore() + }) +}) diff --git a/src/main/window/createMainWindow-renderer-crash-recovery.test.ts b/src/main/window/createMainWindow-renderer-crash-recovery.test.ts index 43dac231130..b3fc210be2c 100644 --- a/src/main/window/createMainWindow-renderer-crash-recovery.test.ts +++ b/src/main/window/createMainWindow-renderer-crash-recovery.test.ts @@ -21,7 +21,7 @@ vi.mock('../browser/browser-client-page-renderer-runtime', async () => { } }) -import { createMainWindow, loadMainWindow } from './createMainWindow' +import { createMainWindow } from './createMainWindow' import { ipcMain } from 'electron' import { shouldRecoverRendererAfterProcessGone } from '../crash-reporting/process-gone-classification' import { @@ -74,8 +74,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -120,8 +120,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -231,8 +231,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -272,8 +272,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -322,8 +322,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -353,6 +353,8 @@ describe('createMainWindow', () => { const windowHandlers: Record void> = {} const webContents = { id: 143, + getURL: vi.fn(() => 'file:///opt/orca/renderer/index.html'), + isDestroyed: vi.fn(() => false), on: vi.fn((event, handler) => { windowHandlers[event] = handler }), @@ -374,8 +376,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -415,10 +417,12 @@ describe('createMainWindow', () => { const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) const { browserWindowInstance, windowHandlers } = createRendererRecoveryWindowHarness() const onBeforeRecoveryReload = vi.fn() + const onRendererRecoveryExhausted = vi.fn() withPlatform('win32', () => { createMainWindow(null, { onBeforeRecoveryReload, + onRendererRecoveryExhausted, shouldRecoverRenderer: (details) => shouldRecoverRendererAfterProcessGone({ reason: details.reason, @@ -436,10 +440,14 @@ describe('createMainWindow', () => { {} as never, { reason: 'killed', exitCode: 1 } as Electron.RenderProcessGoneDetails ) - vi.runAllTimers() + vi.advanceTimersByTime(250) - expect(onBeforeRecoveryReload).toHaveBeenCalledWith(143) + expect(onBeforeRecoveryReload).toHaveBeenCalledWith(143, 'automatic') expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + // Why the watchdog must stay quiet here: this reload is deliberate during logoff, and a process that + // outlives the session-end signal must not put a native modal on screen mid-teardown. + vi.runAllTimers() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() consoleError.mockRestore() }) @@ -615,8 +623,8 @@ describe('createMainWindow', () => { // 1 initial load + 3 recoveries; the 4th crash was refused. expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) - // The recovery prompt's Reload button goes straight to loadMainWindow, which the breaker never gates. - loadMainWindow(browserWindowInstance as unknown as Electron.BrowserWindow) + // The recovery prompt's Reload button takes the watched retry, which the breaker never gates. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(5) // Still-poisoned machine: the next crash re-raises the prompt immediately instead of re-arming auto-reloads. diff --git a/src/main/window/createMainWindow-startup-reveal.test.ts b/src/main/window/createMainWindow-startup-reveal.test.ts index 1a623ab20e3..1200132d319 100644 --- a/src/main/window/createMainWindow-startup-reveal.test.ts +++ b/src/main/window/createMainWindow-startup-reveal.test.ts @@ -57,8 +57,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-system-resume-relay.test.ts b/src/main/window/createMainWindow-system-resume-relay.test.ts index e51f1be0f07..71bf0cd7381 100644 --- a/src/main/window/createMainWindow-system-resume-relay.test.ts +++ b/src/main/window/createMainWindow-system-resume-relay.test.ts @@ -56,8 +56,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return instance diff --git a/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts b/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts index 9b88c60800c..8287b9bcecb 100644 --- a/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts +++ b/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts @@ -51,8 +51,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -143,8 +143,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -237,8 +237,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -302,8 +302,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -366,8 +366,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -424,8 +424,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -486,8 +486,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -565,8 +565,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -630,8 +630,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -707,8 +707,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -777,8 +777,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-tray-minimize-close.test.ts b/src/main/window/createMainWindow-tray-minimize-close.test.ts index 14f829b63d9..469d8313830 100644 --- a/src/main/window/createMainWindow-tray-minimize-close.test.ts +++ b/src/main/window/createMainWindow-tray-minimize-close.test.ts @@ -81,8 +81,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), hide: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return instance diff --git a/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts b/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts index d80349a52df..9d0c3cc9048 100644 --- a/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts +++ b/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts @@ -51,8 +51,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -116,8 +116,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -158,8 +158,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -206,8 +206,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -252,8 +252,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -300,8 +300,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -375,8 +375,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -423,8 +423,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -484,8 +484,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow.test.ts b/src/main/window/createMainWindow.test.ts index 18f4dbf5592..79b9a742533 100644 --- a/src/main/window/createMainWindow.test.ts +++ b/src/main/window/createMainWindow.test.ts @@ -78,8 +78,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -140,8 +140,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -320,8 +320,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) @@ -379,8 +379,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -424,8 +424,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -479,8 +479,8 @@ describe('createMainWindow', () => { }), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -554,8 +554,8 @@ describe('createMainWindow', () => { }), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -630,8 +630,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -678,8 +678,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -726,8 +726,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow.ts b/src/main/window/createMainWindow.ts index 085cc475d49..41d84d1ab93 100644 --- a/src/main/window/createMainWindow.ts +++ b/src/main/window/createMainWindow.ts @@ -14,7 +14,8 @@ import { installMainWindowCloseLifecycle, WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS } from './main-window-close-lifecycle' -import type { CreateMainWindowOptions } from './main-window-contracts' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' +import { mainWindowLoadErrorCode } from './main-window-load-error-code' import { installMainWindowFocusLifecycle } from './main-window-focus-lifecycle' import { installMainWindowShortcutRouting } from './main-window-shortcut-routing' import { installMainWindowStateLifecycle } from './main-window-state-lifecycle' @@ -33,12 +34,25 @@ import { installWindowsPathRegistryChangeListener } from '../pty/windows-path-re export { WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS } -export function loadMainWindow(mainWindow: BrowserWindow): void { - if (is.dev && process.env.ELECTRON_RENDERER_URL) { - void mainWindow.loadURL(process.env.ELECTRON_RENDERER_URL) - } else { - void mainWindow.loadFile(join(__dirname, '../renderer/index.html')) - } +export function loadMainWindow(mainWindow: BrowserWindow, observer?: MainWindowLoadObserver): void { + const load = + is.dev && process.env.ELECTRON_RENDERER_URL + ? mainWindow.loadURL(process.env.ELECTRON_RENDERER_URL) + : mainWindow.loadFile(join(__dirname, '../renderer/index.html')) + // Observe each load promise so failures cannot leave recovery waiting silently. + load.then( + () => observer?.onLoaded?.(), + (cause: unknown) => { + const error = cause instanceof Error ? cause : new Error(String(cause)) + const errorCode = mainWindowLoadErrorCode(error) + // Keep durable diagnostics path-free and exclude shutdown/navigation aborts. + if (!mainWindow.isDestroyed() && errorCode !== 'ERR_ABORTED') { + recordDurableCrashBreadcrumb('main_window_load_failed', { errorCode }) + } + console.error('[window] Main window load failed', error) + observer?.onError?.(error) + } + ) } export function createMainWindow( @@ -158,8 +172,9 @@ export function createMainWindow( } forceRepaint(mainWindow) mainWindow.webContents.send('system:resumed') + // Give a suspended recovery load its full budget on wake. + focus.notifySystemResume() } - powerMonitor.on('resume', onSystemResume) const state = installMainWindowStateLifecycle({ mainWindow, @@ -172,9 +187,11 @@ export function createMainWindow( isWindowClosing: state.isWindowClosing, mainWindow, opts, - reloadMainWindow: () => loadMainWindow(mainWindow), + reloadMainWindow: (observer) => loadMainWindow(mainWindow, observer), rendererWebContentsId }) + // Register after focus is initialized because the resume callback uses it. + powerMonitor.on('resume', onSystemResume) installMainWindowShortcutRouting({ focus, mainWindow, opts, store }) const closeLifecycle = installMainWindowCloseLifecycle({ focus, diff --git a/src/main/window/main-window-contracts.ts b/src/main/window/main-window-contracts.ts index ce5c6cfe0b2..5135be0fbe6 100644 --- a/src/main/window/main-window-contracts.ts +++ b/src/main/window/main-window-contracts.ts @@ -1,4 +1,15 @@ import type { KeybindingOverrides } from '../../shared/keybindings' +import type { + RecoveryExhaustionCause, + RecoveryReloadMilestone, + RecoveryReloadTrigger +} from './renderer-recovery-reload-watchdog' + +/** Per-load outcome from Electron's load promise, which is scoped to that one load unlike `did-finish-load`. */ +export type MainWindowLoadObserver = { + onLoaded?: () => void + onError?: (error: Error) => void +} export type CreateMainWindowOptions = { /** Returns true when a manual app.quit() (Cmd+Q) is in progress, so the renderer skips the running-process confirm dialog. */ @@ -14,11 +25,14 @@ export type CreateMainWindowOptions = { details: Electron.RenderProcessGoneDetails, webContentsId: number ) => boolean - /** Called when consecutive auto-recoveries hit the circuit-breaker limit so the host can prompt instead of crash-looping. */ + /** Called when auto-recovery gives up — the breaker opened, or the recovery reload never produced a document. */ onRendererRecoveryExhausted?: (info: { details: Electron.RenderProcessGoneDetails webContentsId: number recentRecoveryCount: number + cause?: RecoveryExhaustionCause + /** Watched manual retry for the recovery prompt; an unwatched one cannot re-raise the prompt when it stalls too. */ + retry?: () => void }) => void /** Defer renderer load until IPC handlers are registered, or eager renderer calls race into missing channels. */ deferLoad?: boolean @@ -27,6 +41,24 @@ export type CreateMainWindowOptions = { title?: string getKeybindings?: () => KeybindingOverrides | undefined onBeforeReload?: (options: { ignoreCache: boolean; webContentsId: number }) => void - /** Marks the in-place recovery reload so did-finish-load's PTY orphan sweep spares live sessions until restore re-attaches (#5787). */ - onBeforeRecoveryReload?: (webContentsId: number) => void + /** + * Marks the in-place recovery reload so did-finish-load's PTY orphan sweep spares live sessions until restore + * re-attaches (#5787). The prompt's manual Reload is one too, so `trigger` keeps the automatic-recovery + * breadcrumb counting only automatic recoveries. + */ + onBeforeRecoveryReload?: (webContentsId: number, trigger: RecoveryReloadTrigger) => void + /** Pairs an outcome with the recovery-reload intent crumb: bundles could not tell a landed reload from a stalled one. */ + onRecoveryReloadOutcome?: (outcome: { + status: 'loaded' | 'timeout' | 'failed' + attempt: number + elapsedMs: number + /** How far the load got: 'none' is the blank-window field failure, anything else a document that then hung. */ + progress?: RecoveryReloadMilestone + /** True when the load landed after the recovery prompt was already raised — the recovery worked. */ + afterPrompt?: boolean + /** True when a later navigation replaced this load: elapsedMs then measures the replacement, not the reload. */ + superseded?: boolean + /** `ERR_*` code only, for the same reason — Electron's load-error message embeds the URL. */ + errorCode?: string + }) => void } diff --git a/src/main/window/main-window-focus-lifecycle.ts b/src/main/window/main-window-focus-lifecycle.ts index a021d464d15..d494992e692 100644 --- a/src/main/window/main-window-focus-lifecycle.ts +++ b/src/main/window/main-window-focus-lifecycle.ts @@ -14,13 +14,14 @@ import { matchingRichMarkdownContextMenuTableTarget, parseRichMarkdownContextMenuTableTarget } from './editable-context-menu' -import type { CreateMainWindowOptions } from './main-window-contracts' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' import { browserRouteWebContentsRegistry } from '../browser/browser-route-session-runtime' import { attachBrowserClientPageRenderer, retireBrowserClientPageRenderer } from '../browser/browser-client-page-renderer-runtime' import { registerRendererDocumentNavigation } from './renderer-document-navigation' +import { createRendererRecoveryReloadWatchdog } from './renderer-recovery-reload-watchdog' export type MainWindowFocusLifecycle = { dispose: () => void @@ -30,13 +31,15 @@ export type MainWindowFocusLifecycle = { isRendererProcessGone: () => boolean isShortcutRecorderFocused: () => boolean isTerminalInputFocused: () => boolean + /** Relays powerMonitor 'resume' so a suspend-frozen recovery-reload timer does not fire against an unbudgeted load. */ + notifySystemResume: () => void } export function installMainWindowFocusLifecycle(args: { isWindowClosing: () => boolean mainWindow: BrowserWindow opts?: CreateMainWindowOptions - reloadMainWindow: () => void + reloadMainWindow: (observer: MainWindowLoadObserver) => void rendererWebContentsId: number }): MainWindowFocusLifecycle { const { isWindowClosing, mainWindow, opts, reloadMainWindow, rendererWebContentsId } = args @@ -162,6 +165,16 @@ export function installMainWindowFocusLifecycle(args: { rendererRecoveryTimer = null } } + // Why: the reload can stall with a live window and no document — no did-fail-load fires, and the breaker counts + // renderer deaths, so a load that never lands is invisible to every other observer on this path. + const recoveryReloadWatchdog = createRendererRecoveryReloadWatchdog({ + isRecoveryPending: () => rendererRecoveryTimer !== null, + isWindowClosing, + mainWindow, + opts, + reloadMainWindow, + rendererWebContentsId + }) const scheduleRendererRecovery = (details: Electron.RenderProcessGoneDetails): void => { if ( rendererRecoveryTimer || @@ -187,17 +200,16 @@ export function installMainWindowFocusLifecycle(args: { const recovery = rendererRecoveryCircuitBreaker.registerRecoveryAttempt(Date.now()) if (!recovery.allowed) { // Why: too many reloads means it will just crash again; stop and let the host surface a recovery prompt. - opts?.onRendererRecoveryExhausted?.({ - details, - webContentsId: rendererWebContentsId, - recentRecoveryCount: recovery.recentRecoveryCount - }) + // Why through the watchdog: it owns the one-prompt-at-a-time guard, and the prompt's manual retry is a + // recovery reload too — unwatched, one that stalls leaves a blank window and no further prompt. + recoveryReloadWatchdog.escalate( + { details, recentRecoveryCount: recovery.recentRecoveryCount }, + 'crash-loop' + ) return } // Why: a transient renderer/Network Service loss can blank Chromium; reload the app document once to recover. - // Why: mark this in-place reload so the did-finish-load orphan sweep spares live PTYs until session restore (#5787). - opts?.onBeforeRecoveryReload?.(mainWindow.webContents.id) - reloadMainWindow() + recoveryReloadWatchdog.issue(details, recovery.recentRecoveryCount) }, 250) } mainWindow.webContents.on('render-process-gone', (_event, details) => { @@ -229,6 +241,7 @@ export function installMainWindowFocusLifecycle(args: { rendererProcessGone = false attachBrowserClientPageRenderer(rendererWebContents) clearRendererRecoveryTimer() + recoveryReloadWatchdog.notifyDocumentLoaded() }) const dispose = (): void => { @@ -237,6 +250,7 @@ export function installMainWindowFocusLifecycle(args: { resetFloatingTerminalInputFocus() resetShortcutRecorderFocus() clearRendererRecoveryTimer() + recoveryReloadWatchdog.clear() ipcMain.removeListener(markdownFocusChannel, onMarkdownEditorFocused) ipcMain.removeListener(terminalInputFocusChannel, onTerminalInputFocused) ipcMain.removeListener(floatingFocusChannel, onFloatingFocus) @@ -250,6 +264,7 @@ export function installMainWindowFocusLifecycle(args: { isMarkdownEditorFocused: () => markdownEditorFocused, isRendererProcessGone: () => rendererProcessGone, isShortcutRecorderFocused: () => shortcutRecorderFocused, - isTerminalInputFocused: () => terminalInputFocused + isTerminalInputFocused: () => terminalInputFocused, + notifySystemResume: recoveryReloadWatchdog.notifySystemResume } } diff --git a/src/main/window/main-window-load-error-code.ts b/src/main/window/main-window-load-error-code.ts new file mode 100644 index 00000000000..6dd9a79e6d8 --- /dev/null +++ b/src/main/window/main-window-load-error-code.ts @@ -0,0 +1,12 @@ +// Record only the ERR_* code: Electron error messages embed private install URLs. +export function mainWindowLoadErrorCode(error: unknown): string { + const code = + typeof error === 'object' && error !== null && 'code' in error && typeof error.code === 'string' + ? error.code + : undefined + if (code && /^ERR_[A-Z0-9_]+$/.test(code)) { + return code + } + const message = error instanceof Error ? error.message : String(error) + return /\bERR_[A-Z0-9_]+/.exec(message)?.[0] ?? 'unknown' +} diff --git a/src/main/window/renderer-recovery-prompt.test.ts b/src/main/window/renderer-recovery-prompt.test.ts index d5bd7ce11c1..a26700720eb 100644 --- a/src/main/window/renderer-recovery-prompt.test.ts +++ b/src/main/window/renderer-recovery-prompt.test.ts @@ -1,11 +1,14 @@ import type { MessageBoxOptions, MessageBoxReturnValue } from 'electron' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, afterEach, describe, expect, it, vi } from 'vitest' +import { ensureMainI18n, mainI18n } from '../i18n/main-i18n' import type { InstallDirAclPoisonDiagnosis } from '../startup/windows-install-dir-acl-recovery' import { presentRendererRecoveryPrompt, type RendererRecoveryPromptDeps } from './renderer-recovery-prompt' +vi.mock('electron', () => ({ app: { getLocale: () => 'en-US' } })) + const POISON: InstallDirAclPoisonDiagnosis = { detail: "Windows permissions on Orca's install folder are blocking its own sandboxed processes.", commands: ['icacls "C:\\Orca" /grant "*S-1-15-2-2:(OI)(CI)(RX)"', 'icacls "C:\\Orca" /grant b'] @@ -43,17 +46,60 @@ function harness(overrides: Partial & { responses?: } describe('presentRendererRecoveryPrompt', () => { + beforeEach(async () => { + await ensureMainI18n() + await mainI18n.changeLanguage('en') + }) + + afterEach(() => { + mainI18n.removeResourceBundle('en', 'translation') + }) + + it('interpolates the recovery count', async () => { + const { run, shown } = harness({ recentRecoveryCount: 7 }) + await run() + expect(shown[0].detail).toContain('Orca tried to recover 7 times in a row') + expect(shown[0].detail).not.toContain('{{') + }) + + it.each([ + { responses: [1, 0], reloads: 1, quits: 0 }, + { responses: [1, 2], reloads: 0, quits: 1 } + ])( + 'dispatches translated buttons by response index: $responses', + async ({ responses, reloads, quits }) => { + mainI18n.addResourceBundle('en', 'translation', { + rendererRecovery: { reload: 'Recharger', copyCommands: 'Copier', quit: 'Quitter' } + }) + const { run, shown, copied, reload, quit } = harness({ diagnose: () => POISON, responses }) + await run() + expect(shown[0].buttons).toEqual(['Recharger', 'Copier', 'Quitter']) + expect(copied).toEqual([POISON.commands.join('\r\n')]) + expect(reload).toHaveBeenCalledTimes(reloads) + expect(quit).toHaveBeenCalledTimes(quits) + } + ) + it('offers reload and quit with the generic cause when nothing is diagnosed', async () => { const { run, shown, reload, quit } = harness({ responses: [0] }) await run() expect(shown).toHaveLength(1) expect(shown[0].buttons).toEqual(['Reload', 'Quit']) - expect(shown[0].cancelId).toBe(1) + // Escape lands on cancelId, and this box is window-modal over the window it is about: it must not quit. + expect(shown[0].cancelId).toBe(0) expect(shown[0].detail).toContain('graphics-driver or installation problem') expect(reload).toHaveBeenCalledOnce() expect(quit).not.toHaveBeenCalled() }) + it('names the stalled reload instead of claiming a repeated crash', async () => { + const { run, shown } = harness({ failure: 'reload-stalled', responses: [1] }) + await run() + expect(shown[0].message).toContain('stopped responding while reloading') + expect(shown[0].detail).toContain('never finished loading') + expect(shown[0].detail).not.toContain('times in a row') + }) + it('quits on the last button', async () => { const { run, reload, quit } = harness({ responses: [1] }) await run() @@ -65,7 +111,7 @@ describe('presentRendererRecoveryPrompt', () => { const { run, shown } = harness({ diagnose: () => POISON, responses: [0] }) await run() expect(shown[0].buttons).toEqual(['Reload', 'Copy Commands', 'Quit']) - expect(shown[0].cancelId).toBe(2) + expect(shown[0].cancelId).toBe(0) expect(shown[0].detail).toContain(POISON.detail) expect(shown[0].detail).toContain('graphics driver') }) diff --git a/src/main/window/renderer-recovery-prompt.ts b/src/main/window/renderer-recovery-prompt.ts index 2026d1f10b5..18ab02a8eca 100644 --- a/src/main/window/renderer-recovery-prompt.ts +++ b/src/main/window/renderer-recovery-prompt.ts @@ -1,19 +1,13 @@ import type { MessageBoxOptions, MessageBoxReturnValue } from 'electron' +import { translateMain } from '../i18n/main-i18n' import type { InstallDirAclPoisonDiagnosis } from '../startup/windows-install-dir-acl-recovery' +import type { RecoveryExhaustionCause } from './renderer-recovery-reload-watchdog' -/** - * The dialog shown when the renderer crash-loop breaker opens: the window is - * blank by then, so this is the only retry/quit surface the user has. - */ - -const GENERIC_DETAIL = - 'This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.' -// Why keep it alongside the ACL diagnosis: the probe cannot name-check every -// locale, so a driver crash on a healthy install must not lose its only hint. -const DRIVER_FALLBACK = 'If that does not help, the cause is usually a graphics driver.' +export type RendererRecoveryPromptFailure = RecoveryExhaustionCause export type RendererRecoveryPromptDeps = { recentRecoveryCount: number + failure?: RendererRecoveryPromptFailure isQuitting: () => boolean diagnose: () => InstallDirAclPoisonDiagnosis | null showMessageBox: (options: MessageBoxOptions) => Promise @@ -25,29 +19,59 @@ export type RendererRecoveryPromptDeps = { export async function presentRendererRecoveryPrompt( deps: RendererRecoveryPromptDeps ): Promise { - // Why a loop: copying the commands must not dismiss the only surface offering them. + const stalled = deps.failure === 'reload-stalled' + // Copying must preserve the only available recovery surface. while (!deps.isQuitting()) { const diagnosis = deps.diagnose() - const buttons = diagnosis ? ['Reload', 'Copy Commands', 'Quit'] : ['Reload', 'Quit'] + const buttons = [translateMain('rendererRecovery.reload', 'Reload')] + if (diagnosis) { + buttons.push(translateMain('rendererRecovery.copyCommands', 'Copy Commands')) + } + buttons.push(translateMain('rendererRecovery.quit', 'Quit')) + const recoveryDetail = stalled + ? translateMain( + 'rendererRecovery.stalledDetail', + 'Orca reloaded the window after a crash, but it never finished loading.' + ) + : translateMain( + 'rendererRecovery.crashLoopDetail', + 'Orca tried to recover {{recoveryCount}} times in a row without success.', + { recoveryCount: deps.recentRecoveryCount } + ) + const causeDetail = diagnosis + ? `${diagnosis.detail}\n\n${translateMain( + 'rendererRecovery.driverFallback', + 'If that does not help, the cause is usually a graphics driver.' + )}` + : translateMain( + 'rendererRecovery.genericDetail', + 'This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.' + ) const { response } = await deps.showMessageBox({ type: 'error', buttons, defaultId: 0, - cancelId: buttons.length - 1, - title: 'Orca keeps failing to load', - message: 'The app window crashed repeatedly and stopped reloading automatically.', - detail: `Orca tried to recover ${deps.recentRecoveryCount} times in a row without success.\n\n${ - diagnosis ? `${diagnosis.detail}\n\n${DRIVER_FALLBACK}` : GENERIC_DETAIL - }` + // Escape retries instead of destroying the session. + cancelId: 0, + title: translateMain('rendererRecovery.title', 'Orca keeps failing to load'), + message: stalled + ? translateMain( + 'rendererRecovery.stalledMessage', + 'The app window stopped responding while reloading after a crash.' + ) + : translateMain( + 'rendererRecovery.crashLoopMessage', + 'The app window crashed repeatedly and stopped reloading automatically.' + ), + detail: `${recoveryDetail}\n\n${causeDetail}` }) - const choice = buttons[response] - if (choice === 'Copy Commands' && diagnosis) { + if (response === 1 && diagnosis) { deps.copyToClipboard(diagnosis.commands.join('\r\n')) continue } - if (choice === 'Reload') { + if (response === 0) { deps.reload() - } else if (choice === 'Quit') { + } else if (response === buttons.length - 1) { deps.quit() } return diff --git a/src/main/window/renderer-recovery-reload-watchdog.test.ts b/src/main/window/renderer-recovery-reload-watchdog.test.ts new file mode 100644 index 00000000000..b6e854bbbf1 --- /dev/null +++ b/src/main/window/renderer-recovery-reload-watchdog.test.ts @@ -0,0 +1,136 @@ +import { EventEmitter } from 'node:events' +import type { BrowserWindow } from 'electron' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { MainWindowLoadObserver } from './main-window-contracts' +import { + createRendererRecoveryReloadWatchdog, + RENDERER_RECOVERY_LOAD_TIMEOUT_MS +} from './renderer-recovery-reload-watchdog' + +vi.mock('@electron-toolkit/utils', () => ({ is: { dev: false } })) + +function createHarness() { + const webContents = Object.assign(new EventEmitter(), { id: 143 }) + const mainWindow = { webContents, isDestroyed: () => false } as unknown as BrowserWindow + const loads: MainWindowLoadObserver[] = [] + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const watchdog = createRendererRecoveryReloadWatchdog({ + mainWindow, + rendererWebContentsId: webContents.id, + isRecoveryPending: () => false, + isWindowClosing: () => false, + reloadMainWindow: (observer) => loads.push(observer), + opts: { onRecoveryReloadOutcome, onRendererRecoveryExhausted } + }) + const abortLatestLoad = () => loads.at(-1)?.onError?.(new Error('ERR_ABORTED (-3)')) + watchdog.issue({ reason: 'crashed', exitCode: 5 }, 1) + return { + watchdog, + webContents, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } +} + +describe('superseding recovery navigations', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => vi.useRealTimers()) + + it('removes listeners and the pending stall timer during teardown', () => { + const { watchdog, webContents, loads, onRecoveryReloadOutcome } = createHarness() + expect(webContents.eventNames().sort()).toEqual(['did-fail-load', 'did-navigate', 'dom-ready']) + expect(vi.getTimerCount()).toBe(1) + watchdog.clear() + expect(webContents.eventNames()).toEqual([]) + expect(vi.getTimerCount()).toBe(0) + loads[0]?.onLoaded?.() + loads[0]?.onError?.(new Error('ERR_FILE_NOT_FOUND')) + watchdog.notifySystemResume() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(0) + }) + + it('does not mistake a replacement error page for recovery', () => { + const { + watchdog, + webContents, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + abortLatestLoad() + webContents.emit('did-navigate') + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', true) + watchdog.notifyDocumentLoaded() + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'failed', errorCode: 'ERR_FILE_NOT_FOUND' }) + ) + watchdog.clear() + }) + + it('ignores subframe failures and aborted replacement navigations', () => { + const { watchdog, webContents, abortLatestLoad, onRecoveryReloadOutcome } = createHarness() + abortLatestLoad() + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', false) + webContents.emit('did-fail-load', {}, -3, 'ERR_ABORTED', 'file:///previous', true) + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', superseded: true }) + ) + watchdog.clear() + }) + + it('keeps Reload available if the replacement fails beneath an existing prompt', () => { + const { + watchdog, + webContents, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + abortLatestLoad() + webContents.emit('did-navigate') + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', true) + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(loads).toHaveLength(2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + watchdog.clear() + }) + + it('recognizes a successful replacement started after the stall prompt', () => { + const { + watchdog, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + abortLatestLoad() + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', superseded: true, afterPrompt: true }) + ) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(loads).toHaveLength(2) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + watchdog.clear() + }) +}) diff --git a/src/main/window/renderer-recovery-reload-watchdog.ts b/src/main/window/renderer-recovery-reload-watchdog.ts new file mode 100644 index 00000000000..1295ca13ac4 --- /dev/null +++ b/src/main/window/renderer-recovery-reload-watchdog.ts @@ -0,0 +1,310 @@ +import { is } from '@electron-toolkit/utils' +import type { BrowserWindow } from 'electron' +import { isSystemSessionEnding } from '../crash-reporting/expected-teardown-state' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' +import { mainWindowLoadErrorCode } from './main-window-load-error-code' + +// Field recoveries took up to 30.4s; allow 45s before retrying a load with no document. +export const RENDERER_RECOVERY_LOAD_TIMEOUT_MS = 45_000 +// Vite cold starts need a longer budget than packaged files. +export const RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS = 180_000 +// Retry once before handing recovery back to the user. +const RENDERER_RECOVERY_LOAD_ATTEMPTS = 2 +// Milestones may extend the budget, but cannot postpone the prompt indefinitely. +const RENDERER_RECOVERY_LOAD_CAP_FACTOR = 2 + +/** Automatic recovery vs the prompt's manual Reload; they must not share one breadcrumb name. */ +export type RecoveryReloadTrigger = 'automatic' | 'manual-retry' + +/** How far a load got. Ranked, so an attempt's milestone only ever moves forward. */ +export type RecoveryReloadMilestone = 'none' | 'committed' | 'dom-ready' +const MILESTONE_RANK: Record = { + none: 0, + committed: 1, + 'dom-ready': 2 +} + +export type RecoveryExhaustionCause = 'crash-loop' | 'reload-stalled' + +export type RendererRecoveryReloadWatchdog = { + /** Issues a recovery reload and arms the stall watchdog. */ + issue: ( + details: Electron.RenderProcessGoneDetails, + recentRecoveryCount: number, + trigger?: RecoveryReloadTrigger + ) => void + /** Raises the recovery prompt at most once: a native message box cannot be dismissed, so a second one stacks. */ + escalate: (subject: RecoveryPromptSubject, cause: RecoveryExhaustionCause) => void + /** + * A main-frame document finished loading. Only an attempt whose load was superseded takes this as its outcome; + * every other attempt settles through its own load promise, which an error page or a later navigation cannot fool. + */ + notifyDocumentLoaded: () => void + /** Restarts the stall budget after a suspend froze the timer mid-load. */ + notifySystemResume: () => void + clear: () => void +} + +type RecoveryReload = { + attempt: number + details: Electron.RenderProcessGoneDetails + recentRecoveryCount: number + /** Never rewritten: the elapsedMs a crash bundle reads has to stay time-since-issue. */ + issuedAt: number + /** Absolute deadline. A suspend pushes it out; a milestone cannot. */ + capAt: number + milestone: RecoveryReloadMilestone + progressedSinceArm: boolean + /** Chromium aborted this load for a later navigation, which now owns the outcome. */ + superseded: boolean +} + +type RecoveryReloadSeed = Pick +/** What a raised prompt is about; the crash-loop breaker has no attempt to hand over, only the crash. */ +export type RecoveryPromptSubject = Pick + +/** Bounds stalled recovery reloads while still observing success after escalation. */ +export function createRendererRecoveryReloadWatchdog(args: { + /** True when a renderer death has already queued its own recovery, which then owns the next load. */ + isRecoveryPending: () => boolean + isWindowClosing: () => boolean + mainWindow: BrowserWindow + opts?: CreateMainWindowOptions + reloadMainWindow: (observer: MainWindowLoadObserver) => void + rendererWebContentsId: number +}): RendererRecoveryReloadWatchdog { + const { + isRecoveryPending, + isWindowClosing, + mainWindow, + opts, + reloadMainWindow, + rendererWebContentsId + } = args + // Cache before teardown: accessing a destroyed window's webContents throws. + const rendererWebContents = mainWindow.webContents + let inFlight: RecoveryReload | null = null + // Retain timed-out loads so a late success can disarm the prompt's Reload. + let latest: RecoveryReload | null = null + // Keep one prompt until answered; native message boxes cannot be dismissed programmatically. + let prompt: RecoveryPromptSubject | null = null + let documentLanded = false + let timer: ReturnType | null = null + + const clearTimer = (): void => { + if (timer) { + clearTimeout(timer) + timer = null + } + } + // Match loadMainWindow's dev/prod branch. + const timeoutMs = (): number => + is.dev && process.env.ELECTRON_RENDERER_URL + ? RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS + : RENDERER_RECOVERY_LOAD_TIMEOUT_MS + + const armTimer = (reload: RecoveryReload): void => { + clearTimer() + reload.progressedSinceArm = false + timer = setTimeout( + () => onBudgetExpired(reload), + Math.max(0, Math.min(timeoutMs(), reload.capAt - Date.now())) + ) + timer.unref?.() + } + + const onBudgetExpired = (reload: RecoveryReload): void => { + if (inFlight !== reload) { + return + } + // Give a progressing load the remaining budget instead of restarting it cold. + if (reload.progressedSinceArm && Date.now() < reload.capAt) { + armTimer(reload) + return + } + fail(reload) + } + + const start = (seed: RecoveryReloadSeed, trigger: RecoveryReloadTrigger): void => { + const issuedAt = Date.now() + const reload: RecoveryReload = { + ...seed, + issuedAt, + capAt: issuedAt + timeoutMs() * RENDERER_RECOVERY_LOAD_CAP_FACTOR, + milestone: 'none', + progressedSinceArm: false, + superseded: false + } + inFlight = reload + latest = reload + documentLanded = false + // Preserve live PTYs until renderer session restore (#5787). + opts?.onBeforeRecoveryReload?.(mainWindow.webContents.id, trigger) + // Only this load's promise distinguishes success from stale events and error pages. + reloadMainWindow({ + onLoaded: () => settleLoaded(reload), + onError: (error) => onLoadRejected(reload, mainWindowLoadErrorCode(error)) + }) + armTimer(reload) + } + + const settleLoaded = (reload: RecoveryReload): void => { + // A replaced attempt's promise may resolve on the replacement document. + if (reload !== latest) { + return + } + latest = null + documentLanded = true + if (reload === inFlight) { + inFlight = null + clearTimer() + } + opts?.onRecoveryReloadOutcome?.({ + status: 'loaded', + attempt: reload.attempt, + elapsedMs: Math.max(0, Date.now() - reload.issuedAt), + // Record late recovery even if the prompt has already appeared. + ...(prompt ? { afterPrompt: true } : {}), + // Replacement timings must be excluded from recovery-load budget analysis. + ...(reload.superseded ? { superseded: true } : {}) + }) + } + + // ERR_ABORTED transfers ownership to a replacement; the cap still bounds a silent replacement. + const onLoadRejected = (reload: RecoveryReload, errorCode: string): void => { + if (errorCode !== 'ERR_ABORTED') { + fail(reload, errorCode) + return + } + if (latest !== reload) { + return + } + reload.superseded = true + if (inFlight === reload) { + armTimer(reload) + } + } + + const retryFrom = (subject: RecoveryPromptSubject): void => { + prompt = null + // A late recovery makes the prompt's Reload unnecessary. + if (documentLanded) { + return + } + start( + { attempt: 1, details: subject.details, recentRecoveryCount: subject.recentRecoveryCount }, + 'manual-retry' + ) + } + + const escalate = (subject: RecoveryPromptSubject, cause: RecoveryExhaustionCause): void => { + // A new crash invalidates any document that landed while the prompt was open. + documentLanded = false + if (prompt) { + return + } + prompt = subject + opts?.onRendererRecoveryExhausted?.({ + details: subject.details, + webContentsId: rendererWebContentsId, + recentRecoveryCount: subject.recentRecoveryCount, + cause, + // Watch manual retries too, so another stall can offer recovery again. + retry: () => retryFrom(subject) + }) + } + + const fail = (reload: RecoveryReload, errorCode?: string): void => { + // Only the live attempt owns a failure verdict. + if (inFlight !== reload) { + return + } + // Suppress shutdown verdicts; resume may re-arm the retained attempt. + if ( + isWindowClosing() || + opts?.getIsQuitting?.() || + mainWindow.isDestroyed() || + isSystemSessionEnding() + ) { + return + } + inFlight = null + clearTimer() + opts?.onRecoveryReloadOutcome?.({ + status: errorCode === undefined ? 'timeout' : 'failed', + attempt: reload.attempt, + // Wall-clock changes must not produce negative diagnostic durations. + elapsedMs: Math.max(0, Date.now() - reload.issuedAt), + progress: reload.milestone, + ...(errorCode === undefined ? {} : { errorCode }) + }) + // A pending prompt or crash recovery owns the next reload. + if (prompt || isRecoveryPending()) { + return + } + // Restart only loads with no document; preserve progress until the user chooses Reload. + if (reload.attempt < RENDERER_RECOVERY_LOAD_ATTEMPTS && reload.milestone === 'none') { + start({ ...reload, attempt: reload.attempt + 1 }, 'automatic') + return + } + escalate(reload, 'reload-stalled') + } + + // Commit and DOM-ready distinguish a blank load from a document still loading. + const observeMilestone = (milestone: RecoveryReloadMilestone) => (): void => { + if (!inFlight || MILESTONE_RANK[milestone] <= MILESTONE_RANK[inFlight.milestone]) { + return + } + inFlight.milestone = milestone + inFlight.progressedSinceArm = true + } + const onDidNavigate = observeMilestone('committed') + const onDomReady = observeMilestone('dom-ready') + const onDidFailLoad = ( + _event: Electron.Event, + errorCode: number, + errorDescription: string, + _validatedURL: string, + isMainFrame: boolean + ): void => { + if (!isMainFrame || errorCode === -3 || !latest?.superseded) { + return + } + // Error documents also finish loading; only a successful replacement may settle an aborted attempt. + latest.superseded = false + documentLanded = false + fail(latest, mainWindowLoadErrorCode(new Error(errorDescription))) + } + rendererWebContents.on('did-navigate', onDidNavigate) + rendererWebContents.on('dom-ready', onDomReady) + rendererWebContents.on('did-fail-load', onDidFailLoad) + + return { + issue: (details, recentRecoveryCount, trigger = 'automatic') => + start({ attempt: 1, details, recentRecoveryCount }, trigger), + escalate, + notifyDocumentLoaded: () => { + // Timed-out replacements can still recover beneath the prompt. + if (latest?.superseded) { + settleLoaded(latest) + } + }, + // Restore the budget after sleep without rewriting the diagnostic issue time. + notifySystemResume: () => { + if (!inFlight) { + return + } + inFlight.capAt = Date.now() + timeoutMs() * RENDERER_RECOVERY_LOAD_CAP_FACTOR + armTimer(inFlight) + }, + clear: () => { + inFlight = null + latest = null + prompt = null + clearTimer() + rendererWebContents.off?.('did-navigate', onDidNavigate) + rendererWebContents.off?.('dom-ready', onDomReady) + rendererWebContents.off?.('did-fail-load', onDidFailLoad) + } + } +} diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 243771fee8f..f281201638b 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -17590,5 +17590,17 @@ "action": "Try Agents", "hiddenToast": "Agents tab hidden. Re-enable it in Settings → Experimental." } + }, + "rendererRecovery": { + "reload": "Reload", + "copyCommands": "Copy Commands", + "quit": "Quit", + "stalledDetail": "Orca reloaded the window after a crash, but it never finished loading.", + "crashLoopDetail": "Orca tried to recover {{recoveryCount}} times in a row without success.", + "driverFallback": "If that does not help, the cause is usually a graphics driver.", + "genericDetail": "This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.", + "title": "Orca keeps failing to load", + "stalledMessage": "The app window stopped responding while reloading after a crash.", + "crashLoopMessage": "The app window crashed repeatedly and stopped reloading automatically." } } From 36a826ff4875bfb7dfcc3bfba4fe5f79f8d0a625 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:47:32 -0700 Subject: [PATCH 14/26] fix(ssh): compile node-pty from the host's own Node headers instead of nodejs.org (STA-6674) (#18774) * fix(ssh): compile node-pty from the host's own Node headers instead of nodejs.org STA-6674: a Linux SSH host that cannot reach nodejs.org never came up. node-pty ships no Linux prebuild, so npm hands it to node-gyp, and node-gyp's default is to download node-v-headers.tar.gz before configuring. The host refused that connection (ECONNREFUSED) and the relay deploy failed inside npm install, which the UI showed only as "Disconnected". Every official Node build and every version manager that unpacks one already has those exact headers at /include/node. Export node-gyp's nodedir to that prefix, on every command that can compile node-pty (npm install, npm rebuild, the cloexec patch's rebuild), when the shipped node_version.h matches the running Node. Both npm_config_nodedir (node-gyp 10, Node 20) and npm_package_config_node_gyp_nodedir (node-gyp >= 11.4) are set so every Node the relay runs on reads it. A version mismatch leaves it unset, which is the existing behaviour. When a host is both header-less and offline, name that in the deploy error instead of forty lines of gyp http output, with the two remedies. Reproduced and verified with a Docker sshd whose nodejs.org resolves to 127.0.0.1, on node:24.12.0 (the user's version), node:20 and node:26: ssh-relay-offline-node-headers.docker.test.ts. * fix(ssh): fail loudly when node-gyp ignores the exported Node headers dir The headers export relies on npm forwarding npm_config_nodedir / npm_package_config_node_gyp_nodedir into lifecycle scripts. If a future npm drops that, node-gyp would silently fall back to downloading, and an offline host would fail with the same "install an official Node" diagnosis -- wrong, since the host did ship headers. The prefix now echoes ORCA-NODE-HEADERS: into the command's output before the compile, and the download-failure diagnosis reads it back: an exported dir plus a download attempt is reported as an Orca defect naming the dir, not as a host problem. Nothing else changes when it works. * fix(ssh): address review on the relay node-headers export - Unset any inherited npm_config_nodedir / npm_package_config_node_gyp_nodedir before the conditional export, so a stale header dir from the remote profile cannot bypass the version check and build a wrong-ABI binding (CodeRabbit). - Require `gyp ERR! configure error` and a real network errno in the headers-download matcher; node-gyp's fetch client logs retried attempts it recovers from, and a FetchError can be a non-2xx mirror answer (pullfrog). - Say "no local headers matching its own version", since the probe also rejects a version mismatch, not only absent headers (CodeRabbit). - Log the same diagnosis from the non-fatal `npm rebuild` fallback (CodeRabbit). - Docker test waits for the SSH banner on the mapped port before connecting instead of trusting `docker run -d` (CodeRabbit). * fix(ssh): read the node-headers marker from the host output, not the quoted command execCommand rejects with `Command "" failed (exit N): `, and quotes the whole prefix, marker echo included. The first-match parser hit that copy and returned `${ORCA_NODE_HEADERS_DIR:-none}"; ...` as a "dir", so every real no-headers failure was misreported as an Orca defect (measured by an independent Docker exercise of 609685e). Strip the exec- failure head before scanning; keep first-match so gyp output cannot spoof it. The unit fixture hid this by rejecting with `Command "npm install" failed`, a string production never builds. It now rejects from the command the mock actually received, and the Docker test gains a no-headers failure case on the same offline fixture that asserts the host-remedy message. Also unset NPM_CONFIG_NODEDIR (npm accepts either case), and narrow the claim: a ~/.npmrc nodedir= is not overridable from the env (measured: empty env override is ignored on npm 10 and 11), so it stays the operator's setting. Copy the node binary via fs in the unit test so a failed copy fails the test. * docs(ssh): state the header-mismatch refusal as a conservative default, not an observed crash * docs(ssh): note why the exec-failure head regex may match lazily --- src/main/ssh/build-toolchain-diagnosis.ts | 63 ++++++ .../ssh/ssh-relay-build-toolchain.test.ts | 70 ++++++ src/main/ssh/ssh-relay-build-toolchain.ts | 4 +- src/main/ssh/ssh-relay-deploy.ts | 28 ++- ...-native-deps-install-staged-upload.test.ts | 63 ++++++ src/main/ssh/ssh-relay-node-headers.test.ts | 164 ++++++++++++++ src/main/ssh/ssh-relay-node-headers.ts | 97 +++++++++ ...-relay-offline-node-headers.docker.test.ts | 204 ++++++++++++++++++ 8 files changed, 687 insertions(+), 6 deletions(-) create mode 100644 src/main/ssh/ssh-relay-node-headers.test.ts create mode 100644 src/main/ssh/ssh-relay-node-headers.ts create mode 100644 src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts diff --git a/src/main/ssh/build-toolchain-diagnosis.ts b/src/main/ssh/build-toolchain-diagnosis.ts index c64a41ead51..77d20ce475a 100644 --- a/src/main/ssh/build-toolchain-diagnosis.ts +++ b/src/main/ssh/build-toolchain-diagnosis.ts @@ -163,3 +163,66 @@ export function formatMissingToolchainError( ] return lines.join('\n') } + +const NODE_HEADERS_TARBALL_RE = /node-v[0-9.]+-headers\.tar\.gz/i + +/** + * Whether a native-deps failure is node-gyp failing to download Node headers from nodejs.org. + * + * Why it needs naming: the raw output is forty lines of `gyp http` and stack frames around one + * `ECONNREFUSED`, and it reads as a broken host or a broken Orca. Which of two things it is + * depends on what the local-headers export found first, so the formatter takes that answer. + */ +export function isNodeHeadersDownloadFailure(message: string): boolean { + // Why `configure error` is required: node-gyp's fetch client logs `attempt N failed with ` + // on retries it then recovers from, so a network token alone also matches a build that got its + // headers and died later for an unrelated reason. Only the configure step downloads headers. + return ( + /gyp ERR! configure error/i.test(message) && + NODE_HEADERS_TARBALL_RE.test(message) && + /\b(ECONNREFUSED|ENOTFOUND|ETIMEDOUT|EHOSTUNREACH|ENETUNREACH|EAI_AGAIN|ECONNRESET)\b/.test( + message + ) + ) +} + +const NODE_HEADERS_CONTEXT = + 'node-pty has no prebuilt binary for Linux, so it must be compiled on the remote host, and ' + + 'node-gyp fetches the Node.js headers from nodejs.org unless the Node install provides them ' + + 'at /include/node.' + +/** + * @param localHeadersDir what the local-headers export found: a dir it exported, `null` when + * the host's Node ships no matching headers, `undefined` when the answer never came back. + * + * Why the exported-dir case is its own message: the export is the fix, so node-gyp downloading + * anyway means its `nodedir` env keys were not honoured (a future npm dropping the passthrough, + * a wrapper scrubbing the env). That is an Orca defect, not a host problem, and must not be + * reported as one -- it names the dir so the report is checkable. + */ +export function formatNodeHeadersDownloadError( + underlyingError: string, + localHeadersDir: string | null | undefined +): string { + const lines = localHeadersDir + ? [ + `The remote host could not download the Node.js headers needed to compile node-pty, even ` + + `though its Node install ships matching headers at ${localHeadersDir}/include/node and ` + + `Orca pointed node-gyp at them. node-gyp ignored that setting; this is an Orca defect, ` + + `please report it with the log below.`, + '', + 'Workaround on the remote host until then: allow outbound HTTPS to nodejs.org, or point ' + + 'npm at a mirror: npm config set disturl https:///dist' + ] + : [ + 'The remote host could not download the Node.js headers needed to compile node-pty, and ' + + `its Node install has no local headers matching its own version. ${NODE_HEADERS_CONTEXT}`, + '', + 'Fix one of the following on the remote host, then reconnect:', + ' - Install Node.js from an official build or a version manager (nvm, fnm, volta, n), ' + + 'which ship headers for exactly the Node they run; or', + ' - Allow outbound HTTPS to nodejs.org, or point npm at a mirror: ' + + 'npm config set disturl https:///dist' + ] + return [...lines, '', `Underlying install error: ${underlyingError}`].join('\n') +} diff --git a/src/main/ssh/ssh-relay-build-toolchain.test.ts b/src/main/ssh/ssh-relay-build-toolchain.test.ts index b216b599f5a..6fcc4f5321d 100644 --- a/src/main/ssh/ssh-relay-build-toolchain.test.ts +++ b/src/main/ssh/ssh-relay-build-toolchain.test.ts @@ -3,7 +3,9 @@ import { buildToolchainProbeCommand, parseBuildToolchainProbe, formatMissingToolchainError, + formatNodeHeadersDownloadError, formatSkippedNodePtyWarning, + isNodeHeadersDownloadFailure, shouldProbeBuildToolchainAfterNativeDepsFailure } from './ssh-relay-build-toolchain' @@ -125,3 +127,71 @@ describe('formatSkippedNodePtyWarning', () => { expect(warning).toContain('install a C/C++ toolchain') }) }) + +// Verbatim shape of the STA-6674 failure: node-gyp on a host whose nodejs.org is refused. +const HEADERS_REFUSED = + 'npm error gyp http GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz\n' + + 'npm error gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNREFUSED\n' + + 'npm error gyp ERR! configure error\n' + + 'npm error gyp ERR! stack FetchError: request to https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz failed, reason: connect ECONNREFUSED 127.0.0.1:443' + +describe('isNodeHeadersDownloadFailure', () => { + it('matches node-gyp failing to fetch the Node headers tarball', () => { + expect(isNodeHeadersDownloadFailure(HEADERS_REFUSED)).toBe(true) + expect( + isNodeHeadersDownloadFailure( + 'gyp http fetch GET https://nodejs.org/download/release/v20.19.0/node-v20.19.0-headers.tar.gz attempt 1 failed with ENOTFOUND\ngyp ERR! configure error' + ) + ).toBe(true) + }) + + it('is not the toolchain diagnosis, and does not fire on other network failures', () => { + expect(shouldProbeBuildToolchainAfterNativeDepsFailure(HEADERS_REFUSED)).toBe(false) + // The registry, not nodejs.org: a different remedy. + expect( + isNodeHeadersDownloadFailure( + 'npm error network request to https://registry.npmjs.org/node-pty failed, reason: connect ECONNREFUSED' + ) + ).toBe(false) + // Headers named but the build failed for another reason. + expect( + isNodeHeadersDownloadFailure( + 'gyp info using node-v24.12.0-headers.tar.gz\ngyp ERR! build error make failed with exit code: 2' + ) + ).toBe(false) + // A retried attempt that recovered, then a compile failure: not a download failure. + expect( + isNodeHeadersDownloadFailure( + 'gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNRESET\n' + + 'gyp http 200 https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz\n' + + 'gyp ERR! build error\ngyp ERR! stack Error: `make` failed with exit code: 2' + ) + ).toBe(false) + // A mirror answering non-2xx is a FetchError without a network code: a different remedy. + expect( + isNodeHeadersDownloadFailure( + 'gyp ERR! configure error\ngyp ERR! stack FetchError: 404 Not Found https://mirror/dist/v24.12.0/node-v24.12.0-headers.tar.gz' + ) + ).toBe(false) + }) +}) + +describe('formatNodeHeadersDownloadError', () => { + it('names both host remedies when the host ships no headers', () => { + const msg = formatNodeHeadersDownloadError(HEADERS_REFUSED, null) + expect(msg).toContain('no local headers matching its own version') + expect(msg).toContain('/include/node') + expect(msg).toContain('nvm, fnm, volta, n') + expect(msg).toContain('disturl') + expect(msg).toContain('ECONNREFUSED') + }) + + it('reports an Orca defect, not a host problem, when headers were exported and ignored', () => { + const msg = formatNodeHeadersDownloadError(HEADERS_REFUSED, '/usr/local') + expect(msg).toContain('/usr/local/include/node') + expect(msg).toContain('Orca defect') + expect(msg).not.toContain('no local headers matching its own version') + expect(msg).not.toContain('nvm, fnm, volta, n') + expect(msg).toContain('ECONNREFUSED') + }) +}) diff --git a/src/main/ssh/ssh-relay-build-toolchain.ts b/src/main/ssh/ssh-relay-build-toolchain.ts index bcb64d3bdf9..db7b6353697 100644 --- a/src/main/ssh/ssh-relay-build-toolchain.ts +++ b/src/main/ssh/ssh-relay-build-toolchain.ts @@ -16,7 +16,9 @@ export { shouldProbeBuildToolchainAfterNativeDepsFailure, toolchainInstallHintLines, formatSkippedNodePtyWarning, - formatMissingToolchainError + formatMissingToolchainError, + formatNodeHeadersDownloadError, + isNodeHeadersDownloadFailure } from './build-toolchain-diagnosis' export type { BuildToolchainStatus } from './build-toolchain-diagnosis' diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index e7450478d92..5d8101361c6 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -53,11 +53,14 @@ import { } from './ssh-relay-deploy-timing' import { createSshOperationAbortError, shellEscape } from './ssh-connection-utils' import { isWindowsRelayPlatform } from '../../shared/relay-artifacts' +import { exportLocalNodeHeadersPrefix, localNodeHeadersFromOutput } from './ssh-relay-node-headers' import { probeBuildToolchain, formatMissingToolchainError, formatSkippedNodePtyWarning, - shouldProbeBuildToolchainAfterNativeDepsFailure + shouldProbeBuildToolchainAfterNativeDepsFailure, + formatNodeHeadersDownloadError, + isNodeHeadersDownloadFailure } from './ssh-relay-build-toolchain' import { commandWithNodePath, @@ -1174,7 +1177,7 @@ async function installNativeDeps( hostPlatform, nodePath, remoteDir, - `${resetPrefix}npm install --ignore-scripts=false --omit=dev --no-audit --no-fund ${installArgs} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}${resetPrefix}npm install --ignore-scripts=false --omit=dev --no-audit --no-fund ${installArgs} 2>&1` ) await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, @@ -1236,6 +1239,14 @@ async function installNativeDeps( return } } + // Why: either the local-headers export found nothing (a host both header-less and offline) or + // it did and node-gyp downloaded anyway (the export is broken) -- name which, or the log reads + // as a broken relay either way. + if (platform.startsWith('linux') && isNodeHeadersDownloadFailure(msg)) { + throw new Error(formatNodeHeadersDownloadError(msg, localNodeHeadersFromOutput(msg)), { + cause: err + }) + } throw err } @@ -1254,8 +1265,15 @@ async function installNativeDeps( throw err } signal?.throwIfAborted() + // Same diagnosis as the install catch: this fallback is non-fatal, so the log is the only + // place the offline-headers cause can reach anyone. + const rebuildMsg = (err as Error).message console.warn( - `[ssh-relay][NATIVE-DEPS-REBUILD-FAIL] npm rebuild native deps failed at ${remoteDir} (${platform}): ${(err as Error).message}` + `[ssh-relay][NATIVE-DEPS-REBUILD-FAIL] npm rebuild native deps failed at ${remoteDir} (${platform}): ${ + platform.startsWith('linux') && isNodeHeadersDownloadFailure(rebuildMsg) + ? formatNodeHeadersDownloadError(rebuildMsg, localNodeHeadersFromOutput(rebuildMsg)) + : rebuildMsg + }` ) } signal?.throwIfAborted() @@ -1347,7 +1365,7 @@ async function applyNodePtyMasterCloexecPatch( hostPlatform, nodePath, remoteDir, - `${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` ) const output = await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, @@ -1529,7 +1547,7 @@ async function rebuildNativeDeps( hostPlatform, nodePath, remoteDir, - `npm rebuild --ignore-scripts=false ${depNames.map(shellEscape).join(' ')} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}npm rebuild --ignore-scripts=false ${depNames.map(shellEscape).join(' ')} 2>&1` ) await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, diff --git a/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts b/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts index 4d90a7c8289..e8616e7c230 100644 --- a/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts @@ -154,6 +154,69 @@ describe('installNativeDeps staged uploads', () => { expect(writeObservedAt).toBeLessThanOrEqual(npmInstallIdx) }) + it('exports the host Node headers dir to node-gyp on every command that can compile node-pty (STA-6674)', async () => { + const conn = makeMockConnection(sftpCapture) + // Install succeeds, the probe fails, the rebuild repairs it, then the cloexec patch rebuilds again. + feed(makeExecResponses({ npmInstall: 'ok', probe: 'missing', repairProbe: 'ok' })) + + await deployAndLaunchRelay(conn) + + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + const compiling = ['npm install', 'npm rebuild', 'node-pty-1.1.0-master-cloexec-patch.cjs'] + for (const compileStep of compiling) { + const command = commands.find((candidate) => candidate.includes(compileStep)) + expect(command, compileStep).toBeDefined() + // Both spellings: node-gyp 10 (Node 20) reads only npm_config_, node-gyp >= 11.4 prefers the other. + expect(command).toContain('export npm_config_nodedir=') + expect(command).toContain('npm_package_config_node_gyp_nodedir=') + // The export precedes the compile on the same command line, and only when the probe found headers. + expect(command!.indexOf('npm_config_nodedir')).toBeLessThan(command!.indexOf(compileStep)) + expect(command).toContain('node_version.h') + // The marker lands in the captured output, so a failure after it can say what was exported. + expect(command).toContain('echo "ORCA-NODE-HEADERS:${ORCA_NODE_HEADERS_DIR:-none}"') + } + }) + + // What execCommand actually rejects with: the whole command line (marker echo included) quoted + // ahead of the host's output. A fixture that omits the command hides the marker-parsing bug. + function rejectNpmInstallLikeExecCommand(hostOutput: string): void { + vi.mocked(execCommand).mockImplementationOnce(async (_conn, command) => { + throw new Error(`Command "${command}" failed (exit 1): ${hostOutput}`) + }) + } + const HEADERS_REFUSED = + 'npm error gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNREFUSED\nnpm error gyp ERR! configure error' + + it('names the fix when node-gyp cannot download headers and the host ships none (STA-6674)', async () => { + const conn = makeMockConnection(sftpCapture) + feed(makeStagedFirstInstallExecPrefix()) + rejectNpmInstallLikeExecCommand(`ORCA-NODE-HEADERS:none\n${HEADERS_REFUSED}`) + feed(['']) // clean stage root + + const error = await deployAndLaunchRelay(conn).catch((e: Error) => e) + expect((error as Error).message).toContain('could not download the Node.js headers') + expect((error as Error).message).toContain('no local headers matching its own version') + expect((error as Error).message).not.toContain('Orca defect') + expect((error as Error).message).toContain('ECONNREFUSED') + // A full toolchain: the toolchain probe must not run, and this is not a "build tools" error. + expect((error as Error).message).not.toContain('build tools') + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + expect(commands.some((command) => command.includes('command -v "$t"'))).toBe(false) + }) + + it('reports an Orca defect when headers were exported but node-gyp downloaded anyway', async () => { + // The marker says the export happened; a download after it means node-gyp never read the env. + const conn = makeMockConnection(sftpCapture) + feed(makeStagedFirstInstallExecPrefix()) + rejectNpmInstallLikeExecCommand(`ORCA-NODE-HEADERS:/usr/local\n${HEADERS_REFUSED}`) + feed(['']) // clean stage root + + const error = await deployAndLaunchRelay(conn).catch((e: Error) => e) + expect((error as Error).message).toContain('/usr/local/include/node') + expect((error as Error).message).toContain('Orca defect') + expect((error as Error).message).not.toContain('no local headers matching its own version') + }) + it('promotes only after the first-install lock is acquired', async () => { const conn = makeMockConnection(sftpCapture) feed(makeExecResponses({ npmInstall: 'ok', probe: 'ok' })) diff --git a/src/main/ssh/ssh-relay-node-headers.test.ts b/src/main/ssh/ssh-relay-node-headers.test.ts new file mode 100644 index 00000000000..84e9016738b --- /dev/null +++ b/src/main/ssh/ssh-relay-node-headers.test.ts @@ -0,0 +1,164 @@ +import { spawnSync } from 'node:child_process' +import { + chmodSync, + copyFileSync, + mkdtempSync, + mkdirSync, + rmSync, + symlinkSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import process from 'node:process' +import { afterEach, describe, expect, it } from 'vitest' +import { exportLocalNodeHeadersPrefix, localNodeHeadersFromOutput } from './ssh-relay-node-headers' + +const POSIX = process.platform !== 'win32' + +/** Runs the prefix under /bin/sh exactly as the relay does, then prints what node-gyp would see. */ +function runPrefix(nodePath: string): { + nodedir: string + pkgNodedir: string + marker: string | null | undefined +} { + const script = `${exportLocalNodeHeadersPrefix(nodePath)}printf '%s\\n%s\\n' "$npm_config_nodedir" "$npm_package_config_node_gyp_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { encoding: 'utf8' }) + expect(result.status).toBe(0) + const marker = localNodeHeadersFromOutput(result.stdout) + const [nodedir = '', pkgNodedir = ''] = result.stdout + .split('\n') + .filter((line) => !line.startsWith('ORCA-NODE-HEADERS:')) + return { nodedir, pkgNodedir, marker } +} + +/** A fake `/bin/node` whose `include/node/node_version.h` claims `version`. */ +function fakeNodePrefix(root: string, version: string): string { + const prefix = join(root, 'prefix') + mkdirSync(join(prefix, 'bin'), { recursive: true }) + mkdirSync(join(prefix, 'include', 'node'), { recursive: true }) + const [major, minor, patch] = version.split('.') + writeFileSync( + join(prefix, 'include', 'node', 'node_version.h'), + `#define NODE_MAJOR_VERSION ${major}\n#define NODE_MINOR_VERSION ${minor}\n#define NODE_PATCH_VERSION ${patch}\n` + ) + // Why a symlink to the real binary: the probe reads process.execPath, which Node resolves + // through symlinks -- so this stands in for `/usr/bin/node -> /opt/node/bin/node` shims too. + symlinkSync(process.execPath, join(prefix, 'bin', 'node')) + return join(prefix, 'bin', 'node') +} + +describe.skipIf(!POSIX)('exportLocalNodeHeadersPrefix', () => { + const roots: string[] = [] + afterEach(() => { + for (const root of roots.splice(0)) { + rmSync(root, { recursive: true, force: true }) + } + }) + + it('exports nodedir when the running Node ships headers for its own version', () => { + // The test runner's Node is an official build, so its prefix has include/node. + const prefix = dirname(dirname(process.execPath)) + const { nodedir, pkgNodedir, marker } = runPrefix(process.execPath) + expect(nodedir).toBe(prefix) + expect(pkgNodedir).toBe(prefix) + expect(marker).toBe(prefix) + }) + + it('leaves nodedir unset when the shipped headers are for another Node version', () => { + // A symlinked node resolves execPath to the real binary, whose prefix is the real one; so + // to stage a mismatch the probe must run a node whose execPath lands in the fake prefix. + // A copy does that. + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const prefix = join(root, 'prefix') + mkdirSync(join(prefix, 'bin'), { recursive: true }) + mkdirSync(join(prefix, 'include', 'node'), { recursive: true }) + writeFileSync( + join(prefix, 'include', 'node', 'node_version.h'), + '#define NODE_MAJOR_VERSION 1\n#define NODE_MINOR_VERSION 0\n#define NODE_PATCH_VERSION 0\n' + ) + const copied = join(prefix, 'bin', 'node') + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const { nodedir, pkgNodedir, marker } = runPrefix(copied) + expect(nodedir).toBe('') + expect(pkgNodedir).toBe('') + expect(marker).toBeNull() + }) + + it('leaves nodedir unset when the prefix has no headers at all', () => { + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const copied = join(root, 'bin', 'node') + mkdirSync(dirname(copied), { recursive: true }) + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const { nodedir } = runPrefix(copied) + expect(nodedir).toBe('') + }) + + it('follows a symlinked node to the install that owns the headers', () => { + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const shim = fakeNodePrefix(root, '0.0.0') + // The shim's own fake headers are ignored: execPath resolves to the real binary, and the + // real prefix's headers are the ones that match. + const { nodedir } = runPrefix(shim) + expect(nodedir).toBe(dirname(dirname(process.execPath))) + }) + + it('clears an inherited nodedir when the probe finds no matching headers', () => { + // A remote profile's stale nodedir must not survive past the version check. + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const copied = join(root, 'bin', 'node') + mkdirSync(dirname(copied), { recursive: true }) + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const script = `${exportLocalNodeHeadersPrefix(copied)}printf '%s|%s|%s' "$npm_config_nodedir" "$NPM_CONFIG_NODEDIR" "$npm_package_config_node_gyp_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { + encoding: 'utf8', + env: { + ...process.env, + npm_config_nodedir: '/usr/stale-headers', + NPM_CONFIG_NODEDIR: '/usr/stale-headers', + npm_package_config_node_gyp_nodedir: '/usr/stale-headers' + } + }) + expect(result.status).toBe(0) + expect(result.stdout.split('\n').at(-1)).toBe('||') + }) + + it('does not fail the command line when node itself cannot run', () => { + const script = `${exportLocalNodeHeadersPrefix('/nonexistent/node')}echo "after:$npm_config_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { encoding: 'utf8' }) + expect(result.status).toBe(0) + expect(result.stdout.trim()).toBe('ORCA-NODE-HEADERS:none\nafter:') + }) +}) + +describe('localNodeHeadersFromOutput', () => { + it('reads the host answer, not the copy of the marker echo quoted in an exec-failure head', () => { + // The real shape: execCommand quotes the whole command line, prefix included, before the output. + const command = `export PATH='/usr/local/bin':$PATH && cd '/root/.orca-remote/relay-x' && ${exportLocalNodeHeadersPrefix('/usr/local/bin/node')}npm install node-pty 2>&1` + const failed = (hostOutput: string): string => + `Command "${command}" failed (exit 1): ${hostOutput}` + expect( + localNodeHeadersFromOutput(failed('ORCA-NODE-HEADERS:none\ngyp ERR! configure error')) + ).toBeNull() + expect( + localNodeHeadersFromOutput(failed('ORCA-NODE-HEADERS:/usr/local\ngyp ERR! configure error')) + ).toBe('/usr/local') + // No host output at all after the head: the command copy alone must not count as a marker. + expect(localNodeHeadersFromOutput(failed(''))).toBeUndefined() + }) + + it('distinguishes an exported dir, an explicit none, and no marker at all', () => { + expect(localNodeHeadersFromOutput('x\nORCA-NODE-HEADERS:/usr/local\ngyp ERR!')).toBe( + '/usr/local' + ) + expect(localNodeHeadersFromOutput('ORCA-NODE-HEADERS:none\ngyp ERR!')).toBeNull() + expect(localNodeHeadersFromOutput('gyp ERR! only')).toBeUndefined() + }) +}) diff --git a/src/main/ssh/ssh-relay-node-headers.ts b/src/main/ssh/ssh-relay-node-headers.ts new file mode 100644 index 00000000000..a590bd40fca --- /dev/null +++ b/src/main/ssh/ssh-relay-node-headers.ts @@ -0,0 +1,97 @@ +/** + * Point node-gyp at the headers the host's Node install already ships, so compiling node-pty + * needs nothing from nodejs.org. + * + * Why: node-pty has no Linux prebuild, so every Linux relay compiles it, and node-gyp's default + * is to download `node-v-headers.tar.gz` before configuring. Every official Node build, and + * every version manager that unpacks one (nvm, fnm, volta, mise, n), already has those exact + * headers at `/include/node`. The download was the only step that needed the internet, + * so a firewalled host failed with ECONNREFUSED on work that never had to happen (STA-6674). + * + * Why both variables: node-gyp >= 11.4 prefers `npm_package_config_node_gyp_` and npm 11+ + * warns that arbitrary `npm_config_` is deprecated, but node-gyp 10 (bundled with Node 20) + * reads only `npm_config_`. Both together cover every Node the relay runs on. + * + * Why the version check: node-gyp trusts `nodedir` blindly, so a distro `/usr/include/node` left + * by an older headers package would be compiled against as-is. Whether that binding then misbehaves + * is not established (one measured run loaded a node-20-header build under node 24); refusing is + * the conservative default. A mismatch leaves the variables unset, which is today's path. + */ +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable the probe answers into; namespaced so it cannot collide with npm's own. */ +const NODEDIR_SHELL_VAR = 'ORCA_NODE_HEADERS_DIR' + +/** + * Prints the running Node's install prefix when `/include/node/node_version.h` matches + * `process.versions.node`, and nothing otherwise. `process.execPath` is symlink-resolved, so a + * `/usr/bin/node` -> `/opt/node/bin/node` shim still finds `/opt/node/include`. + */ +export const LOCAL_NODE_HEADERS_PROBE_JS = [ + 'const p=require("path"),f=require("fs");', + 'const d=p.dirname(p.dirname(process.execPath));', + 'try{', + 'const h=f.readFileSync(p.join(d,"include","node","node_version.h"),"utf8");', + 'const v=["MAJOR","MINOR","PATCH"].map(k=>(h.match(new RegExp("#define NODE_"+k+"_VERSION ([0-9]+)"))||[])[1]).join(".");', + 'if(v===process.versions.node)process.stdout.write(d)', + '}catch{}' +].join('') + +/** + * Stdout marker naming what the probe found, printed before the compile so the answer is in the + * captured output of any failure that follows. `none` means no matching local headers. + */ +export const LOCAL_NODE_HEADERS_MARKER_PREFIX = 'ORCA-NODE-HEADERS:' + +/** + * POSIX-sh prefix (`...; `) that exports node-gyp's `nodedir` for the rest of the command line + * when the host's Node ships matching headers. Prepend to any command that may compile node-pty: + * `npm install`, `npm rebuild`, and the cloexec patch (its `npm rebuild` inherits the env). + */ +export function exportLocalNodeHeadersPrefix(nodePath: string): string { + const probe = `${shellEscape(nodePath)} -e ${shellEscape(LOCAL_NODE_HEADERS_PROBE_JS)} 2>/dev/null` + // Why the unset: a remote profile can already export a nodedir (a stale distro header dir), in + // either case npm accepts. Left alone it would bypass the version check above and compile + // against those headers. Deliberately env only: a `nodedir=` in ~/.npmrc is not reachable from here + // -- npm ignores an empty env override, and a CLI `--nodedir=` would also override the good + // export -- so an npmrc setting stays the operator's, as it was before this prefix existed. + return ( + `${NODEDIR_SHELL_VAR}=$(${probe}); ` + + `unset npm_config_nodedir NPM_CONFIG_NODEDIR npm_package_config_node_gyp_nodedir; ` + + `if [ -n "$${NODEDIR_SHELL_VAR}" ]; then ` + + `export npm_config_nodedir="$${NODEDIR_SHELL_VAR}" npm_package_config_node_gyp_nodedir="$${NODEDIR_SHELL_VAR}"; ` + + `fi; ` + + `echo "${LOCAL_NODE_HEADERS_MARKER_PREFIX}\${${NODEDIR_SHELL_VAR}:-none}"; ` + ) +} + +/** + * The headers dir the prefix exported, `null` when it found none, or `undefined` when the + * marker is absent (output truncated, or the command never reached the prefix). + */ +export function localNodeHeadersFromOutput(output: string): string | null | undefined { + // Why the head is stripped first: a failed exec's message is `Command "" failed + // (exit N): `, and quotes this prefix verbatim -- including the marker's + // `echo`. Scanning from the start would match that copy and return `${ORCA_NODE_HEADERS_DIR:- + // none}"...` as a "dir". Only what follows the head is the host's answer. + const head = output.match(EXEC_FAILURE_HEAD_RE) + const hostOutput = head ? output.slice(head[0].length) : output + // First match, not last: the host's own line comes first, and later lines are npm/gyp output + // that must not be able to spoof it. + for (const line of hostOutput.split(/\r?\n/)) { + const at = line.indexOf(LOCAL_NODE_HEADERS_MARKER_PREFIX) + if (at === -1) { + continue + } + const dir = line.slice(at + LOCAL_NODE_HEADERS_MARKER_PREFIX.length).trim() + return dir === 'none' || dir === '' ? null : dir + } + return undefined +} + +/** + * `Command "" failed (exit N): ` -- see ssh-relay-exec-command.ts. + * Lazy `[\s\S]*?` is safe: it stops at the first `" failed (exit N): `, and no command this + * module builds contains that literal, so the match cannot end early inside the command. + */ +const EXEC_FAILURE_HEAD_RE = /^Command "[\s\S]*?" failed \(exit -?\d+\): / diff --git a/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts b/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts new file mode 100644 index 00000000000..c5700c991c1 --- /dev/null +++ b/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts @@ -0,0 +1,204 @@ +// Why this exists (STA-6674): a Linux host whose only unreachable endpoint is nodejs.org could +// not run a relay. node-pty ships no Linux prebuild, so npm hands it to node-gyp, and node-gyp +// downloads `node-v-headers.tar.gz` unless told the host already has the headers -- which +// every official Node install does, at `/include/node`. This drives the real deploy at a +// Docker sshd whose nodejs.org resolves to 127.0.0.1 (ECONNREFUSED, exactly what the user saw). +// +// Run: ORCA_REVIEW_SSH_OFFLINE_HEADERS=1 pnpm test src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts +// Needs Docker and `pnpm build:relay`. ORCA_REVIEW_SSH_NODE_IMAGE picks the Node image +// (default node:24.12.0-bookworm, the user's version); ORCA_REVIEW_SSH_TARGET_HOST overrides +// the address the app connects to (default 127.0.0.1). +import { execFileSync, spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { connect } from 'node:net' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ app: { getAppPath: () => process.cwd() } })) + +import { SshConnection } from './ssh-connection' +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import type { SshTarget } from '../../shared/ssh-types' + +const RUN_REVIEW_ORACLE = process.env.ORCA_REVIEW_SSH_OFFLINE_HEADERS === '1' +const NODE_IMAGE = process.env.ORCA_REVIEW_SSH_NODE_IMAGE ?? 'node:24.12.0-bookworm' +const TARGET_HOST = process.env.ORCA_REVIEW_SSH_TARGET_HOST ?? '127.0.0.1' + +type TargetFixture = { + containerName: string + identityFile: string + port: number + tempDir: string +} + +function run(command: string, args: string[], timeout = 30_000, input?: string): string { + return execFileSync(command, args, { + encoding: 'utf8', + stdio: [input === undefined ? 'ignore' : 'pipe', 'pipe', 'pipe'], + timeout, + input + }).trim() +} + +function dockerExec(fixture: TargetFixture, command: string): string { + return run('docker', ['exec', fixture.containerName, 'bash', '-lc', command], 60_000) +} + +async function startTarget(): Promise { + const image = `orca-review-offline-headers:${NODE_IMAGE.replace(/[^A-Za-z0-9_.-]/g, '-')}` + run( + 'docker', + ['build', '-q', '-t', image, '-'], + 600_000, + [ + `FROM ${NODE_IMAGE}`, + 'RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends openssh-server git && rm -rf /var/lib/apt/lists/* && mkdir -p /run/sshd /root/.ssh && chmod 700 /root/.ssh', + '' + ].join('\n') + ) + const tempDir = mkdtempSync(join(tmpdir(), 'orca-offline-headers-ssh-')) + const identityFile = join(tempDir, 'id_ed25519') + run('ssh-keygen', ['-t', 'ed25519', '-N', '', '-f', identityFile, '-q']) + const publicKey = readFileSync(`${identityFile}.pub`, 'utf8').trim() + const containerName = `orca-offline-headers-${randomUUID().slice(0, 12)}` + // Why a refused connection and not a dropped one: a timeout takes node-gyp's retry path and + // burns the deploy budget; the user's host refused, and that is the path under test. + run( + 'docker', + [ + 'run', + '-d', + '--name', + containerName, + '--add-host', + 'nodejs.org:127.0.0.1', + '-p', + '0.0.0.0::22', + '-e', + `AUTHORIZED_KEY=${publicKey}`, + image, + 'bash', + '-lc', + 'printf "%s\\n" "$AUTHORIZED_KEY" > /root/.ssh/authorized_keys && chmod 600 /root/.ssh/authorized_keys && exec /usr/sbin/sshd -D -e' + ], + 120_000 + ) + const port = Number(run('docker', ['port', containerName, '22/tcp']).split(':').at(-1)) + // `docker run -d` returns before sshd binds; connect() against a closed port is a flake. + await waitForSshBanner(port) + return { containerName, identityFile, port, tempDir } +} + +/** Resolves once sshd answers with its banner on the mapped port, or throws after the deadline. */ +async function waitForSshBanner(port: number, deadlineMs = 60_000): Promise { + const deadline = Date.now() + deadlineMs + for (;;) { + const gotBanner = await new Promise((resolve) => { + const socket = connect({ host: TARGET_HOST, port }) + const done = (value: boolean): void => { + socket.destroy() + resolve(value) + } + socket.setTimeout(2_000, () => done(false)) + socket.once('data', (chunk) => done(chunk.toString('utf8').startsWith('SSH-'))) + socket.once('error', () => done(false)) + }) + if (gotBanner) { + return + } + if (Date.now() > deadline) { + throw new Error(`sshd on port ${port} did not answer within ${deadlineMs / 1000}s`) + } + await new Promise((resolve) => setTimeout(resolve, 500)) + } +} + +function stopTarget(fixture: TargetFixture | null): void { + if (!fixture) { + return + } + spawnSync('docker', ['rm', '-f', fixture.containerName], { stdio: 'ignore', timeout: 30_000 }) + rmSync(fixture.tempDir, { recursive: true, force: true }) +} + +function createConnection(fixture: TargetFixture): SshConnection { + const target: SshTarget = { + id: `offline-headers-${randomUUID()}`, + label: 'Offline node headers Docker SSH target', + source: 'manual', + host: TARGET_HOST, + port: fixture.port, + username: 'root', + identityFile: fixture.identityFile, + identitiesOnly: true + } + return new SshConnection(target, { onStateChange: vi.fn() }) +} + +describe.skipIf(!RUN_REVIEW_ORACLE)( + 'SSH relay deploy on a host that cannot reach nodejs.org', + () => { + let fixture: TargetFixture | null = null + + beforeAll(async () => { + fixture = await startTarget() + }, 900_000) + + afterAll(() => { + stopTarget(fixture) + }) + + it('compiles node-pty from the host Node install headers instead of downloading them', async () => { + const activeFixture = fixture as TargetFixture + expect(dockerExec(activeFixture, 'getent hosts nodejs.org')).toContain('127.0.0.1') + const connection = createConnection(activeFixture) + await connection.connect() + try { + const result = await deployAndLaunchRelay(connection, undefined, 60) + expect(result.remoteRelayDir).toBeTruthy() + + const evidence = dockerExec( + activeFixture, + [ + `cd '${result.remoteRelayDir}'`, + 'test -f node_modules/node-pty/build/Release/pty.node && echo PTY_NODE=built', + 'test -d /root/.cache/node-gyp && echo HEADERS=downloaded || echo HEADERS=local', + `node -e "require('node-pty'); require('@parcel/watcher'); console.log('NATIVE=loadable')"` + ].join('; ') + ) + console.log(`[offline-node-headers] ${NODE_IMAGE}: ${evidence.replace(/\n/g, ' ')}`) + expect(evidence).toContain('PTY_NODE=built') + expect(evidence).toContain('HEADERS=local') + expect(evidence).toContain('NATIVE=loadable') + } finally { + await connection.disconnect() + } + }, 600_000) + + it('names the missing-local-headers cause, not an Orca defect, when the host ships no headers', async () => { + // Same offline host, headers removed and the relay uninstalled so the deploy compiles again. + // This is the shape a review found misreported: the exec-failure message quotes the whole + // command (marker echo included) ahead of the output, and the parser must not read that copy. + const activeFixture = fixture as TargetFixture + dockerExec( + activeFixture, + 'rm -rf /usr/local/include/node /root/.orca-remote /root/.cache/node-gyp' + ) + const connection = createConnection(activeFixture) + await connection.connect() + try { + const error = await deployAndLaunchRelay(connection, undefined, 60).catch((e: Error) => e) + expect(error).toBeInstanceOf(Error) + const message = (error as Error).message + console.log(`[offline-node-headers] ${NODE_IMAGE} no-headers: ${message.split('\n')[0]}`) + expect(message).toContain('no local headers matching its own version') + expect(message).not.toContain('Orca defect') + expect(message).toContain('ECONNREFUSED') + } finally { + await connection.disconnect() + } + }, 600_000) + } +) From 821c8b7df0894e58b475f8a91b282089904ade8d Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:50:46 -0700 Subject: [PATCH 15/26] Bump mobile app.json to 0.0.48 (#18801) Co-authored-by: Merge Sim --- mobile/app.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mobile/app.json b/mobile/app.json index d5a420cc74b..131e3899396 100644 --- a/mobile/app.json +++ b/mobile/app.json @@ -2,7 +2,7 @@ "expo": { "name": "Orca", "slug": "orca-mobile", - "version": "0.0.47", + "version": "0.0.48", "orientation": "default", "icon": "./assets/icon.png", "userInterfaceStyle": "automatic", From 7856e5a677c8f6891a7f67e0fe0d704ee1453a75 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:55:54 -0700 Subject: [PATCH 16/26] fix(sidebar): confirm filter reset before revealing active workspace (#18708) * fix(sidebar): confirm filter reset before revealing active workspace * Polish workspace reveal confirmation and focus primary action * Fix reveal confirmation CI: commit-phase ref and sidebar test provider --- .../components/confirmation-dialog-context.ts | 4 + .../components/confirmation-dialog.test.tsx | 31 +++ .../src/components/confirmation-dialog.tsx | 58 ++++-- .../WorktreeList.card-memo-stability.test.tsx | 4 + ....lineage-agent-expansion-coupling.test.tsx | 4 + ...ktreeList.lineage-child-real-card.test.tsx | 4 + ...treeList.status-lane-lineage-drop.test.tsx | 4 + ...worktree-list-lineage-card-test-harness.ts | 13 +- .../navigation/use-reveal-requests.test.tsx | 191 ++++++++++++++++++ .../navigation/use-reveal-requests.ts | 53 ++++- src/renderer/src/i18n/locales/en.json | 8 + 11 files changed, 350 insertions(+), 24 deletions(-) create mode 100644 src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.test.tsx diff --git a/src/renderer/src/components/confirmation-dialog-context.ts b/src/renderer/src/components/confirmation-dialog-context.ts index 4675c181fb7..b112191a4e6 100644 --- a/src/renderer/src/components/confirmation-dialog-context.ts +++ b/src/renderer/src/components/confirmation-dialog-context.ts @@ -1,4 +1,5 @@ import { createContext, useContext } from 'react' +import type { LucideIcon } from 'lucide-react' // Keep the context component-free so Fast Refresh preserves its identity. @@ -9,6 +10,9 @@ export type ConfirmationDialogOptions = { confirmLabel?: string cancelLabel?: string confirmVariant?: 'default' | 'destructive' + icon?: LucideIcon + cancelVariant?: 'outline' | 'ghost' + initialFocus?: 'confirm' /** Renders a "Don't ask again" checkbox. `onConfirmed` runs only when the user confirms with it checked. */ dontAskAgain?: { label?: string; onConfirmed: () => void } } diff --git a/src/renderer/src/components/confirmation-dialog.test.tsx b/src/renderer/src/components/confirmation-dialog.test.tsx index 25738ee4c17..5097a5d7786 100644 --- a/src/renderer/src/components/confirmation-dialog.test.tsx +++ b/src/renderer/src/components/confirmation-dialog.test.tsx @@ -44,6 +44,37 @@ function renderDialog(options: ConfirmationDialogOptions): { onSettled: ReturnTy describe('ConfirmationDialogProvider', () => { afterEach(cleanup) + it('focuses the primary action when requested and confirms with Enter', async () => { + const { onSettled } = renderDialog({ + title: 'Reveal hidden workspace?', + confirmLabel: 'Clear filters and reveal', + initialFocus: 'confirm' + }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => + expect(screen.getByRole('button', { name: 'Clear filters and reveal' })).toHaveFocus() + ) + await userEvent.keyboard('{Enter}') + await waitFor(() => expect(onSettled).toHaveBeenCalledWith(true)) + }) + + it('still cancels with Escape when the primary action has focus', async () => { + const { onSettled } = renderDialog({ + title: 'Reveal hidden workspace?', + initialFocus: 'confirm' + }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => expect(screen.getByRole('button', { name: 'Confirm' })).toHaveFocus()) + await userEvent.keyboard('{Escape}') + await waitFor(() => expect(onSettled).toHaveBeenCalledWith(false)) + }) + + it('keeps the default cancel focus for callers that do not opt in', async () => { + renderDialog({ title: 'Delete artifact?', confirmVariant: 'destructive' }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => expect(screen.getByRole('button', { name: 'Cancel' })).toHaveFocus()) + }) + it('omits the checkbox unless the caller opts in', async () => { renderDialog({ title: 'Delete artifact?' }) diff --git a/src/renderer/src/components/confirmation-dialog.tsx b/src/renderer/src/components/confirmation-dialog.tsx index a1670b9bb22..615a2400bc6 100644 --- a/src/renderer/src/components/confirmation-dialog.tsx +++ b/src/renderer/src/components/confirmation-dialog.tsx @@ -32,6 +32,7 @@ export function ConfirmationDialogProvider({ children: React.ReactNode }): React.JSX.Element { const nextIdRef = useRef(0) + const confirmButtonRef = useRef(null) const [queue, setQueue] = useState([]) const [dontAskAgain, setDontAskAgain] = useState(false) const activeRequest = queue[0] ?? null @@ -46,6 +47,7 @@ export function ConfirmationDialogProvider({ } // Why: Radix keeps dialog content mounted while closing; keep labels stable without a post-render Effect. const displayedRequest = activeRequest ?? lastDisplayedRequestRef.current + const Icon = displayedRequest?.options.icon useEffect(() => { // Why: this provider's dialog is not represented by activeModal. Block @@ -96,18 +98,37 @@ export function ConfirmationDialogProvider({ open={activeRequest !== null} onOpenChange={(open) => !open && settleActiveRequest(false)} > - - - {displayedRequest?.options.title} - {displayedRequest?.options.description ? ( - // Callers pass multi-line descriptions (e.g. one path per line). - - {displayedRequest.options.description} - - ) : null} - + { + if (activeRequest?.options.initialFocus === 'confirm') { + event.preventDefault() + confirmButtonRef.current?.focus() + } + }} + > +
+ {Icon && ( +
+
+ )} + + {displayedRequest?.options.title} + {displayedRequest?.options.description ? ( + // Callers pass multi-line descriptions (e.g. one path per line). + + {displayedRequest.options.description} + + ) : null} + +
{displayedRequest?.options.dontAskAgain ? (
) : null} - - + + + {expanded ? ( +
+ {props.tasks.length > 0 ? ( +
    + {props.tasks.map((task) => ( +
  • +
  • + ))} +
+ ) : ( +

+ {translate( + 'components.native-chat.backgroundTasks.detailsUnavailable', + 'Task details are unavailable for this session.' + )} +

+ )} +
+ ) : null} + + + ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx index 356a482dfa3..afdf5ace7c5 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx @@ -4,6 +4,7 @@ import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-libra import React, { forwardRef, useImperativeHandle } from 'react' import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' import { decodeAgentSessionQuestionAnswers } from '../../../../shared/agent-session-question-answer' import type { NativeChatQuestionCardProps } from './NativeChatQuestionCard' @@ -17,13 +18,19 @@ const mocks = vi.hoisted(() => ({ showTurnStatus?: boolean runtimeContext?: unknown }, - composerProps: null as null | { structuredTransport?: Record }, + composerProps: null as null | { + structuredTransport?: Record + isWorking?: boolean + }, questionCardProps: null as NativeChatQuestionCardProps | null, promptItems: [] as AgentJournalRenderItem[], respond: vi.fn(), handlePasteEvent: vi.fn(), pasteFromClipboard: vi.fn(), - submissions: [] as unknown[] + submissions: [] as unknown[], + monitoringBackgroundTasks: false, + backgroundTasks: [] as AgentSessionBackgroundTask[], + stopBackgroundTasks: vi.fn() })) vi.mock('@/runtime/structured-agent-session-client', () => ({ @@ -67,8 +74,11 @@ vi.mock('./use-structured-agent-session', async () => { send: outbox.send, retry: outbox.retry, isWorking: false, + isMonitoringBackgroundTasks: mocks.monitoringBackgroundTasks, + backgroundTasks: mocks.backgroundTasks, turnId: null, cancel: vi.fn(), + stopBackgroundTasks: mocks.stopBackgroundTasks, respond: mocks.respond, optionSnapshot: [ { @@ -155,6 +165,9 @@ describe('NativeChatStructuredSession', () => { mocks.handlePasteEvent.mockReset() mocks.pasteFromClipboard.mockReset() mocks.submissions = [] + mocks.monitoringBackgroundTasks = false + mocks.stopBackgroundTasks.mockReset() + mocks.backgroundTasks = [] }) it('routes app-menu paste into the structured composer', () => { @@ -210,6 +223,47 @@ describe('NativeChatStructuredSession', () => { } ) + it('places background monitoring above the usable composer and stops without an active turn', async () => { + mocks.monitoringBackgroundTasks = true + mocks.backgroundTasks = [ + { id: 'task-command', kind: 'command', description: 'sleep 180' }, + { id: 'task-agent', kind: 'agent' } + ] + mocks.stopBackgroundTasks.mockResolvedValue({ cancelled: true }) + + render( + + ) + + const status = screen + .getByText('Monitoring background tasks') + .closest('[data-native-chat-background-tasks="true"]') + const composer = screen.getByTestId('structured-composer') + if (!status) { + throw new Error('background task status was not rendered') + } + expect(status.compareDocumentPosition(composer) & Node.DOCUMENT_POSITION_FOLLOWING).toBeTruthy() + expect(mocks.composerProps?.isWorking).toBe(false) + expect(screen.queryByRole('list', { name: 'Running background tasks' })).toBeNull() + + const disclosure = screen.getByRole('button', { name: 'Monitoring background tasks' }) + expect(disclosure.getAttribute('aria-expanded')).toBe('false') + fireEvent.click(disclosure) + expect(disclosure.getAttribute('aria-expanded')).toBe('true') + expect(screen.getByRole('list', { name: 'Running background tasks' })).toBeTruthy() + expect(screen.getByText('sleep 180')).toBeTruthy() + expect(screen.getByText('Background agent')).toBeTruthy() + + fireEvent.click(screen.getByRole('button', { name: 'Stop' })) + await waitFor(() => expect(mocks.stopBackgroundTasks).toHaveBeenCalledOnce()) + }) + it('routes a bare model command to the native option picker', async () => { render( (null) + const [stoppingBackgroundTasks, setStoppingBackgroundTasks] = useState(false) const [optionPickerRequest, setOptionPickerRequest] = useState<{ id: string sequence: number @@ -272,6 +274,16 @@ export function NativeChatStructuredSession( {controller.error ?? composerError}

) : null} + {controller.isMonitoringBackgroundTasks ? ( + { + setStoppingBackgroundTasks(true) + void controller.stopBackgroundTasks().finally(() => setStoppingBackgroundTasks(false)) + }} + /> + ) : null} {prompt ? null : ( { if (!isVisible || !optionCatalog) { @@ -241,8 +243,15 @@ export function useStructuredAgentSession(args: { send: outboxController.send, retry: outboxController.retry, isWorking: turnId !== null, + isMonitoringBackgroundTasks, + backgroundTasks: state.backgroundTasks?.tasks ?? [], turnId, cancel: (turnId: string) => mutate('agentSession.cancel', 'agentSession.cancel', { turnId }), + stopBackgroundTasks: () => + mutate('agentSession.cancel', 'agentSession.cancel', { + turnId: 'background-tasks', + scope: 'background-tasks' + }), respond: (item: StructuredPromptItem, optionId: string) => mutate( item.body.kind === 'approval' diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 9791e83ddf0..12d500608f5 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16986,6 +16986,17 @@ "workedFor": "Worked for {{value0}}", "toggleDetails": "Toggle turn details" }, + "backgroundTasks": { + "monitoring": "Monitoring background tasks", + "stop": "Stop", + "agent": "Background agent", + "workflow": "Background workflow", + "command": "Background command", + "monitor": "Background monitor", + "task": "Background task", + "runningList": "Running background tasks", + "detailsUnavailable": "Task details are unavailable for this session." + }, "jumpToLatest": "Jump to latest", "toggle": { "showTerminal": "Show terminal", diff --git a/src/shared/agent-session-wire.ts b/src/shared/agent-session-wire.ts index 6c2b3791024..40a408df899 100644 --- a/src/shared/agent-session-wire.ts +++ b/src/shared/agent-session-wire.ts @@ -49,6 +49,18 @@ export type AgentSessionHandoffRequest = { export type AgentSessionHandoffResult = { status: AgentSessionHandoffStatus } +export type AgentSessionBackgroundTask = { + id: string + kind: 'agent' | 'workflow' | 'command' | 'monitor' | 'unknown' + description?: string +} + +export type AgentSessionBackgroundTaskState = { + state: 'monitoring' + /** Optional so mixed-version clients can consume state-only hosts. */ + tasks?: AgentSessionBackgroundTask[] +} + /** Backward paging is the client's normal read; 40 matches the page size the * mobile list renders without a visible fill-in. */ export const AGENT_SESSION_HISTORY_DEFAULT_LIMIT = 40 @@ -92,6 +104,8 @@ export type AgentSessionHistoryPage = { liveCursor?: AgentJournalCursor hasOlder: boolean hasNewer: boolean + /** Present on hosts that expose provider-owned background task lifecycle. */ + backgroundTasks?: AgentSessionBackgroundTaskState | null } export type AgentSessionHistoryResult = @@ -123,6 +137,7 @@ export type AgentSessionSubscribeEvent = page: AgentSessionHistoryPage fence: number handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null } | { type: 'batch' @@ -131,6 +146,7 @@ export type AgentSessionSubscribeEvent = /** Added with handoff state so mixed-version cursors retain the ownership fence. */ fence?: number handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null } | { type: 'reset' @@ -139,6 +155,7 @@ export type AgentSessionSubscribeEvent = page: AgentSessionHistoryPage fence: number handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null } | { type: 'end' } diff --git a/src/shared/structured-agent-session-coalescer.test.ts b/src/shared/structured-agent-session-coalescer.test.ts new file mode 100644 index 00000000000..770b08308af --- /dev/null +++ b/src/shared/structured-agent-session-coalescer.test.ts @@ -0,0 +1,56 @@ +import { describe, expect, it } from 'vitest' +import type { AgentSessionSubscribeEvent } from './agent-session-wire' +import { createStructuredAgentSessionEventCoalescer } from './structured-agent-session-coalescer' + +function batch( + sequence: number, + backgroundTasks?: Extract['backgroundTasks'] +): Extract { + return { + type: 'batch', + sessionId: 'session-1', + batch: { + cursor: { epoch: 'epoch-1', sequence }, + items: [], + removedItemIds: [], + submissions: [] + }, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + } +} + +describe('structured agent session event coalescer', () => { + it('preserves background task state when a journal batch follows it', () => { + const events: AgentSessionSubscribeEvent[] = [] + const coalescer = createStructuredAgentSessionEventCoalescer((event) => events.push(event)) + + coalescer.push( + batch(1, { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + }) + ) + coalescer.push(batch(2)) + coalescer.flush() + + expect(events).toHaveLength(1) + expect(events[0]).toMatchObject({ + backgroundTasks: { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + } + }) + }) + + it('keeps an explicit terminal state as the newest coalesced value', () => { + const events: AgentSessionSubscribeEvent[] = [] + const coalescer = createStructuredAgentSessionEventCoalescer((event) => events.push(event)) + + coalescer.push(batch(1, { state: 'monitoring' })) + coalescer.push(batch(1, null)) + coalescer.flush() + + expect(events).toHaveLength(1) + expect(events[0]).toMatchObject({ backgroundTasks: null }) + }) +}) diff --git a/src/shared/structured-agent-session-coalescer.ts b/src/shared/structured-agent-session-coalescer.ts index 51bc7fa0537..fe982d67a69 100644 --- a/src/shared/structured-agent-session-coalescer.ts +++ b/src/shared/structured-agent-session-coalescer.ts @@ -35,7 +35,15 @@ function mergeBatch( ...(right.fence !== undefined || left.fence !== undefined ? { fence: right.fence ?? left.fence } : {}), - ...(right.handoff || left.handoff ? { handoff: right.handoff ?? left.handoff } : {}) + ...(right.handoff || left.handoff ? { handoff: right.handoff ?? left.handoff } : {}), + ...(right.backgroundTasks !== undefined || left.backgroundTasks !== undefined + ? { + backgroundTasks: + right.backgroundTasks !== undefined + ? right.backgroundTasks + : (left.backgroundTasks ?? null) + } + : {}) } } diff --git a/src/shared/structured-agent-session-reducer.test.ts b/src/shared/structured-agent-session-reducer.test.ts index 222db53a564..99ced770641 100644 --- a/src/shared/structured-agent-session-reducer.test.ts +++ b/src/shared/structured-agent-session-reducer.test.ts @@ -250,4 +250,129 @@ describe('structured agent session reducer', () => { expect(state.submissions[0]?.clientMessageId).toBe('client-44') expect(state.submissions.at(-1)?.clientMessageId).toBe('client-299') }) + + it('projects additive background task state without changing transcript identity', () => { + const initial = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: null + } + }) + const monitoring = reduceStructuredAgentSession(initial, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: initial.cursor!, + items: [], + removedItemIds: [], + submissions: [] + }, + fence: 1, + backgroundTasks: { state: 'monitoring' } + } + }) + + expect(monitoring.backgroundTasks).toEqual({ state: 'monitoring' }) + expect(monitoring.items).toBe(initial.items) + }) + + it('returns the same state for duplicate background task publications', () => { + const monitoring = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: { state: 'monitoring' } + } + }) + const duplicate = reduceStructuredAgentSession(monitoring, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: monitoring.cursor!, + items: [], + removedItemIds: [], + submissions: [] + }, + fence: 1, + backgroundTasks: { state: 'monitoring' } + } + }) + + expect(duplicate).toBe(monitoring) + }) + + it('applies background task roster changes without a journal update', () => { + const monitoring = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'first command' }] + } + } + }) + const changed = reduceStructuredAgentSession(monitoring, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: monitoring.cursor!, + items: [], + removedItemIds: [], + submissions: [] + }, + fence: 1, + backgroundTasks: { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'agent', description: 'review the change' }] + } + } + }) + + expect(changed).not.toBe(monitoring) + expect(changed.backgroundTasks?.tasks).toEqual([ + { id: 'task-1', kind: 'agent', description: 'review the change' } + ]) + expect(changed.items).toBe(monitoring.items) + }) + + it('clears additive background state when a replacement snapshot omits the field', () => { + const monitoring = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: { state: 'monitoring' } + } + }) + const withoutCapability = reduceStructuredAgentSession(monitoring, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 2, + page: hydrationPage([item('message', 1)]) + } + }) + + expect(withoutCapability.backgroundTasks).toBeUndefined() + }) }) diff --git a/src/shared/structured-agent-session-reducer.ts b/src/shared/structured-agent-session-reducer.ts index 48029b5910e..0a330539548 100644 --- a/src/shared/structured-agent-session-reducer.ts +++ b/src/shared/structured-agent-session-reducer.ts @@ -4,6 +4,7 @@ import type { AgentJournalSubmission } from './agent-session-journal-types' import type { + AgentSessionBackgroundTaskState, AgentSessionHandoffStatus, AgentSessionHistoryPage, AgentSessionSubscribeEvent @@ -19,6 +20,7 @@ export type StructuredAgentSessionState = { status: 'idle' | 'loading' | 'ready' | 'error' error?: string handoff: AgentSessionHandoffStatus | null + backgroundTasks?: AgentSessionBackgroundTaskState | null } export type StructuredAgentSessionAction = @@ -42,10 +44,35 @@ export const EMPTY_STRUCTURED_AGENT_SESSION: StructuredAgentSessionState = { const MAX_RETAINED_SUBMISSIONS = 256 +function backgroundTaskStatesEqual( + left: AgentSessionBackgroundTaskState | null | undefined, + right: AgentSessionBackgroundTaskState | null | undefined +): boolean { + if (left === right) { + return true + } + if (!left || !right || left.state !== right.state) { + return false + } + if (left.tasks === right.tasks) { + return true + } + if (!left.tasks || !right.tasks || left.tasks.length !== right.tasks.length) { + return false + } + return left.tasks.every( + (task, index) => + task.id === right.tasks?.[index]?.id && + task.kind === right.tasks[index]?.kind && + task.description === right.tasks[index]?.description + ) +} + function replacePage( page: AgentSessionHistoryPage, fence: number, - handoff?: AgentSessionHandoffStatus + handoff?: AgentSessionHandoffStatus, + backgroundTasks?: AgentSessionBackgroundTaskState | null ): StructuredAgentSessionState { return { epoch: page.epoch, @@ -55,7 +82,12 @@ function replacePage( submissions: page.submissions, hasOlder: page.hasOlder, status: 'ready', - handoff: handoff ?? null + handoff: handoff ?? null, + ...(backgroundTasks !== undefined + ? { backgroundTasks } + : page.backgroundTasks !== undefined + ? { backgroundTasks: page.backgroundTasks } + : {}) } } @@ -113,12 +145,23 @@ export function reduceStructuredAgentSession( state.cursor && (!pageCursor || pageCursor.sequence <= state.cursor.sequence) ) { + const backgroundTasksChanged = + action.page.backgroundTasks !== undefined && + !backgroundTaskStatesEqual(action.page.backgroundTasks, state.backgroundTasks) if ( pageCursor?.sequence === state.cursor.sequence && - action.page.fence !== undefined && - action.page.fence !== state.fence + ((action.page.fence !== undefined && action.page.fence !== state.fence) || + backgroundTasksChanged) ) { - return { ...state, fence: action.page.fence, status: 'ready', error: undefined } + return { + ...state, + ...(action.page.fence !== undefined ? { fence: action.page.fence } : {}), + ...(action.page.backgroundTasks !== undefined + ? { backgroundTasks: action.page.backgroundTasks } + : {}), + status: 'ready', + error: undefined + } } return state } @@ -133,7 +176,12 @@ export function reduceStructuredAgentSession( : action.page.submissions, hasOlder: action.page.hasOlder, status: 'ready', - handoff: state.handoff + handoff: state.handoff, + ...(action.page.backgroundTasks !== undefined + ? { backgroundTasks: action.page.backgroundTasks } + : state.backgroundTasks !== undefined + ? { backgroundTasks: state.backgroundTasks } + : {}) } } if (action.type === 'older-page') { @@ -152,7 +200,7 @@ export function reduceStructuredAgentSession( return state } if (event.type === 'snapshot' || event.type === 'reset') { - return replacePage(event.page, event.fence, event.handoff) + return replacePage(event.page, event.fence, event.handoff, event.backgroundTasks) } if (state.epoch !== event.batch.cursor.epoch) { return state @@ -160,15 +208,37 @@ export function reduceStructuredAgentSession( if (state.cursor && event.batch.cursor.sequence < state.cursor.sequence) { return state } + const backgroundTasks = + event.backgroundTasks !== undefined ? event.backgroundTasks : state.backgroundTasks + const journalUnchanged = + event.batch.items.length === 0 && + event.batch.removedItemIds.length === 0 && + event.batch.submissions.length === 0 + if ( + event.batch.cursor.sequence === state.cursor?.sequence && + journalUnchanged && + (event.fence === undefined || event.fence === state.fence) && + (event.handoff === undefined || event.handoff === state.handoff) && + backgroundTaskStatesEqual(backgroundTasks, state.backgroundTasks) && + state.status === 'ready' && + state.error === undefined + ) { + return state + } return { ...state, cursor: event.batch.cursor, fence: event.fence ?? state.fence, - items: mergeItems(state.items, event.batch.items, event.batch.removedItemIds), - submissions: mergeSubmissions(state.submissions, event.batch.submissions), + items: journalUnchanged + ? state.items + : mergeItems(state.items, event.batch.items, event.batch.removedItemIds), + submissions: journalUnchanged + ? state.submissions + : mergeSubmissions(state.submissions, event.batch.submissions), status: 'ready', error: undefined, - handoff: event.handoff ?? state.handoff + handoff: event.handoff ?? state.handoff, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) } } From 6dc7e9fb6a7958009a2a11f18e96ea3fc79c6280 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:22:31 -0700 Subject: [PATCH 22/26] =?UTF-8?q?fix(sidebar):=20stop=20=E2=8C=98=E2=87=A7?= =?UTF-8?q?=E2=86=93=20worktree=20navigation=20jumping=20to=20the=20first?= =?UTF-8?q?=20row=20(#18804)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../sidebar/worktree-keyboard-cycle.test.ts | 7 +- .../sidebar/worktree-keyboard-cycle.ts | 50 ++++++- .../use-keyboard.host-identity.test.tsx | 122 ++++++++++++++++++ .../worktree-list/navigation/use-keyboard.ts | 33 +++-- 4 files changed, 187 insertions(+), 25 deletions(-) create mode 100644 src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx diff --git a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts index c4c4bda6da1..430736c92be 100644 --- a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts +++ b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts @@ -165,8 +165,11 @@ describe('WorktreeList keyboard cycling', () => { // Why: a second buildRows call drifts from the rendered layout (host sections, // pinned placement); cycling must read the same rows the viewport renders. - expect(navigateWorktree).toContain('getCyclableWorktrees(rows, pinnedDisplayPolicy)') - expect(navigateWorktree).toContain('getWorktreeHostIdentity') + expect(navigateWorktree).toContain('getCyclableWorktreeRows(rows, pinnedDisplayPolicy)') + expect(navigateWorktree).toContain('getCyclableRowIdentity') + // Why: the active host is stored resolved while a local row is unqualified; comparing raw identities wraps to the top. + expect(navigateWorktree).toContain('resolveActiveCycleIdentity') + expect(navigateWorktree).not.toContain('composeWorktreeHostIdentity') expect(navigateWorktree).toContain('executionHostId: nextWorktree.hostId') expect(navigateWorktree).toContain('resolveCycledWorktreeId') expect(navigateWorktree).not.toContain('buildRows(') diff --git a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts index 0a2bf23e905..4c50c78b5f5 100644 --- a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts +++ b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts @@ -1,9 +1,49 @@ import type { HostSectionRow } from './host-section-rows' import type { Worktree } from '../../../../shared/worktree/types' -import { getWorktreeHostIdentity } from '../../../../shared/worktree/host-qualified-identity' +import { composeWorktreeHostIdentity } from '../../../../shared/worktree/host-qualified-identity' +import { getWorktreeExecutionHostId, type ExecutionHostId } from '../../../../shared/execution-host' import type { PinnedWorktreeDisplayPolicy, WorktreeRow } from './worktree-list/grouping/row-types' import { getPreferredWorktreeRows } from './worktree-sidebar-row-preference' +/** Host-resolved identity for a cyclable row. + * + * Why resolved rather than `getWorktreeHostIdentity`: a local worktree carries no + * `hostId` (`withRepoHostOwnership` leaves it unqualified), but every activation + * path stores the host it resolved to, so raw and resolved identities never match. + */ +export function getCyclableRowIdentity(row: Pick): string { + return composeWorktreeHostIdentity( + getWorktreeExecutionHostId(row.worktree, row.repo), + row.worktree.id + ) +} + +export function getCyclableWorktreeRows( + rows: readonly HostSectionRow[], + pinnedDisplayPolicy: PinnedWorktreeDisplayPolicy +): WorktreeRow[] { + const itemRows = rows.filter((row): row is WorktreeRow => row.type === 'item') + return getPreferredWorktreeRows(itemRows, pinnedDisplayPolicy) +} + +/** Identity that locates the active workspace among the cyclable rows. */ +export function resolveActiveCycleIdentity(args: { + rows: readonly WorktreeRow[] + activeWorktreeId: string | null + activeWorkspaceExecutionHostId: ExecutionHostId | null +}): string | null { + const { rows, activeWorktreeId, activeWorkspaceExecutionHostId } = args + if (!activeWorktreeId) { + return null + } + if (activeWorkspaceExecutionHostId) { + return composeWorktreeHostIdentity(activeWorkspaceExecutionHostId, activeWorktreeId) + } + // Host-unqualified activation names no host; the row it landed on does. + const row = rows.find((candidate) => candidate.worktree.id === activeWorktreeId) + return row ? getCyclableRowIdentity(row) : null +} + /** Worktree ids in sidebar order, taken from the rows the sidebar actually * rendered, so collapsed groups and collapsed host sections drop out on their own. */ export function getCyclableWorktreeIds( @@ -12,11 +52,10 @@ export function getCyclableWorktreeIds( ): string[] { // Why item-only: folder workspaces render as their own row type and are not // activatable through activateAndRevealWorktree, so cycling has never included them. - const itemRows = rows.filter((row): row is WorktreeRow => row.type === 'item') const ids: string[] = [] const seen = new Set() - for (const row of getPreferredWorktreeRows(itemRows, pinnedDisplayPolicy)) { - const identity = getWorktreeHostIdentity(row.worktree) + for (const row of getCyclableWorktreeRows(rows, pinnedDisplayPolicy)) { + const identity = getCyclableRowIdentity(row) if (seen.has(identity)) { continue } @@ -30,8 +69,7 @@ export function getCyclableWorktrees( rows: readonly HostSectionRow[], pinnedDisplayPolicy: PinnedWorktreeDisplayPolicy ): Worktree[] { - const itemRows = rows.filter((row): row is WorktreeRow => row.type === 'item') - return getPreferredWorktreeRows(itemRows, pinnedDisplayPolicy).map((row) => row.worktree) + return getCyclableWorktreeRows(rows, pinnedDisplayPolicy).map((row) => row.worktree) } /** Pick the worktree that `worktree.navigateUp` / `worktree.navigateDown` moves diff --git a/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx new file mode 100644 index 00000000000..efac59a9c07 --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx @@ -0,0 +1,122 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Worktree } from '../../../../../../shared/worktree/types' +import type { HostSectionRow } from '../../host-section-rows' +import type { RenderRow } from '../listing/render-row' +import { getShortcutPlatform } from '@/lib/shortcut-platform' + +const activateAndRevealWorktree = vi.fn() + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorktree: (...args: unknown[]) => activateAndRevealWorktree(...args) +})) + +vi.mock('@/store', () => ({ + useAppStore: (selector: (state: { keybindings: undefined }) => unknown) => + selector({ keybindings: undefined }) +})) + +const { useWorktreeListKeyboardNavigation } = await import('./use-keyboard') + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +const repo = { id: 'repo-1', path: '/repo-1', displayName: 'Repo 1' } + +// Local worktrees carry no `hostId` — `withRepoHostOwnership` leaves them unqualified. +function localRow(id: string): HostSectionRow & { type: 'item' } { + return { + type: 'item', + rowKey: `row:${id}`, + sectionKey: 'repo:repo-1', + worktree: { id, repoId: repo.id } as unknown as Worktree, + repo: repo as never, + depth: 0, + groupDepth: 0, + lineageTrail: [], + isLastLineageChild: false, + lineageChildCount: 0 + } +} + +const rows: HostSectionRow[] = [localRow('a'), localRow('b'), localRow('c')] +const renderRows = rows as unknown as RenderRow[] + +let container: HTMLDivElement +let root: Root + +function press(direction: 'up' | 'down'): void { + const mod = getShortcutPlatform() === 'darwin' ? { metaKey: true } : { ctrlKey: true } + act(() => { + window.dispatchEvent( + new KeyboardEvent('keydown', { + key: direction === 'down' ? 'ArrowDown' : 'ArrowUp', + code: direction === 'down' ? 'ArrowDown' : 'ArrowUp', + shiftKey: true, + bubbles: true, + cancelable: true, + ...mod + }) + ) + }) +} + +function renderProbe(activeWorktreeId: string, activeHostId: 'local' | null): void { + function Probe(): null { + useWorktreeListKeyboardNavigation({ + rows, + renderRows, + activeWorktreeId, + activeWorkspaceExecutionHostId: activeHostId, + pinnedDisplayPolicy: 'single-location', + virtualizer: { scrollToIndex: () => {} } as never, + scrollRef: { current: null }, + activeModal: 'none', + markDirectScrollInput: () => {} + }) + return null + } + act(() => root.render()) +} + +beforeEach(() => { + activateAndRevealWorktree.mockClear() + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() +}) + +describe('worktree keyboard cycling with a resolved active host', () => { + it('steps to the next row when the active host resolved to local but rows are unqualified', () => { + // Why: a sidebar click activates with the repo-resolved host (`local`), while + // local rows carry no hostId; a raw identity compare misses and wraps to the top. + renderProbe('b', 'local') + + press('down') + + expect(activateAndRevealWorktree).toHaveBeenCalledWith('c', {}) + }) + + it('steps to the previous row when the active host resolved to local', () => { + renderProbe('b', 'local') + + press('up') + + expect(activateAndRevealWorktree).toHaveBeenCalledWith('a', {}) + }) + + it('still steps normally when the active host is unqualified', () => { + renderProbe('b', null) + + press('down') + + expect(activateAndRevealWorktree).toHaveBeenCalledWith('c', {}) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts index 5746e16bf2a..c8d3ec5d80d 100644 --- a/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts +++ b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts @@ -4,16 +4,17 @@ import type { Virtualizer } from '@tanstack/react-virtual' import { useAppStore } from '@/store' import { activateAndRevealWorktree } from '@/lib/worktree-activation' import type { ExecutionHostId } from '../../../../../../shared/execution-host' -import { - composeWorktreeHostIdentity, - getWorktreeHostIdentity -} from '../../../../../../shared/worktree/host-qualified-identity' import { getShortcutPlatform } from '@/lib/shortcut-platform' import { keybindingMatchesAction } from '../../../../../../shared/keybindings' import type { HostSectionRow } from '../../host-section-rows' import type { PinnedWorktreeDisplayPolicy } from '../grouping/row-types' import type { RenderRow } from '../listing/render-row' -import { getCyclableWorktrees, resolveCycledWorktreeId } from '../../worktree-keyboard-cycle' +import { + getCyclableRowIdentity, + getCyclableWorktreeRows, + resolveActiveCycleIdentity, + resolveCycledWorktreeId +} from '../../worktree-keyboard-cycle' import { findPreferredRenderRowIndexForWorktreeIdentity } from './render-row-lookup' function isEditableTarget(target: EventTarget | null): boolean { @@ -65,24 +66,22 @@ export function useWorktreeListKeyboardNavigation(args: { // Why: cycle over the rows the sidebar actually rendered — collapsing a group // means "not now", and a rebuilt near-copy would drift from what is on screen // (host sections, pinned placement, folder workspaces). - const worktrees = getCyclableWorktrees(rows, pinnedDisplayPolicy) - const worktreeIdentities = worktrees.map(getWorktreeHostIdentity) + const worktreeRows = getCyclableWorktreeRows(rows, pinnedDisplayPolicy) const nextWorktreeIdentity = resolveCycledWorktreeId({ - worktreeIds: worktreeIdentities, - activeWorktreeId: activeWorktreeId - ? composeWorktreeHostIdentity( - activeWorkspaceExecutionHostId ?? undefined, - activeWorktreeId - ) - : null, + worktreeIds: worktreeRows.map(getCyclableRowIdentity), + activeWorktreeId: resolveActiveCycleIdentity({ + rows: worktreeRows, + activeWorktreeId, + activeWorkspaceExecutionHostId + }), direction }) if (nextWorktreeIdentity === null) { return } - const nextWorktree = worktrees.find( - (worktree) => getWorktreeHostIdentity(worktree) === nextWorktreeIdentity - ) + const nextWorktree = worktreeRows.find( + (row) => getCyclableRowIdentity(row) === nextWorktreeIdentity + )?.worktree if (!nextWorktree) { return } From 4c5077d57a4c4c1ee591655ded479df042a45998 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:34:36 -0700 Subject: [PATCH 23/26] perf(persistence): skip rewriting unchanged terminal scrollback snapshots (#18764) * perf(terminal): tighten the partial-escape-tail benchmark and equivalence test * perf(terminal): spell the ESC gate the same way as the sibling ingest gates * test(terminal): differential-fuzz the ESC-free partial-escape-tail gate against the unguarded fold * test(terminal): make the escape-tail fuzz exhaustive at symbol depth, and cap the fold expectation Two review findings on the differential fuzz, both about the test faithfully modelling the function it guards. The odometer generated strings by symbol depth but the caller filtered on `chunk.length`, which is the UTF-16 code-unit count. An astral symbol is two code units, so every depth-4 string containing one was silently skipped and the corpus was not exhaustive at depth 4 the way the test name claimed. The generator now yields `{ depth, text }` and the caller filters on depth. That restores the missing strings and takes the pinned corpus from 516,566 to 593,468 - exactly the count CodeRabbit derived for the intended corpus. The pairing assertion in the sibling suite compared the capped `advancePartialEscapeTail` against an uncapped `extractPartialEscapeTail(pending + chunk)`. It passed only because no pairing in that corpus crosses MAX_PARTIAL_ESCAPE_TAIL_LENGTH; it would have stopped modelling the function the moment one did. The cap now lives in the expectation, matching the fuzz oracle. Re-verified the fuzz still fails on a wrong guard: mutating the gate to a bracket check fails all four tests with a `gate diverged` assertion on a lone ESC chunk. Reported by CodeRabbit and pullfrog on #18748. --- ...terminal-partial-escape-tail-benchmark.mjs | 158 ++++++------------ package.json | 1 + .../terminal-partial-escape-tail.fuzz.test.ts | 154 +++++++++++++++++ .../terminal-partial-escape-tail.test.ts | 53 +++--- src/shared/terminal-partial-escape-tail.ts | 2 +- 5 files changed, 229 insertions(+), 139 deletions(-) create mode 100644 src/shared/terminal-partial-escape-tail.fuzz.test.ts diff --git a/config/scripts/terminal-partial-escape-tail-benchmark.mjs b/config/scripts/terminal-partial-escape-tail-benchmark.mjs index 01acbf95bd6..0337daa4e9d 100644 --- a/config/scripts/terminal-partial-escape-tail-benchmark.mjs +++ b/config/scripts/terminal-partial-escape-tail-benchmark.mjs @@ -1,147 +1,87 @@ #!/usr/bin/env node -// Benchmarks the partial-escape-tail fold that runs once per PTY chunk, on the main thread, for -// every terminal. Drives the production export against a baseline reproducing the pre-change -// shape (unconditional concat + per-code-unit walk), and proves equivalence over a fuzz corpus -// before timing so the gate cannot silently change what the tracker returns. -import { spawnSync } from 'node:child_process' +// Times the partial-escape-tail fold that runs once per PTY chunk for every terminal against a +// baseline with the pre-change shape (unconditional concat + per-code-unit walk). Equivalence is +// proven over a corpus first, so the reported speedup cannot come from the gate changing the answer. import { performance } from 'node:perf_hooks' -import fs from 'node:fs' -import nodeModule from 'node:module' -import path from 'node:path' -import process from 'node:process' -import { fileURLToPath } from 'node:url' +import { + advancePartialEscapeTail, + extractPartialEscapeTail, + MAX_PARTIAL_ESCAPE_TAIL_LENGTH +} from '../../src/shared/terminal-partial-escape-tail.ts' -if (!process.execArgv.includes('--experimental-transform-types')) { - const result = spawnSync( - process.execPath, - ['--experimental-transform-types', '--no-warnings', import.meta.filename], - { stdio: 'inherit' } - ) - process.exit(result.status ?? 1) -} +const CHUNK_BYTES = 16 * 1024 +const CHUNKS = 640 +const ROUNDS = 7 -nodeModule.registerHooks({ - resolve(specifier, context, nextResolve) { - if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) { - const candidate = new URL(`${specifier}.ts`, context.parentURL) - if (fs.existsSync(fileURLToPath(candidate))) { - return { url: candidate.href, shortCircuit: true } - } - } - return nextResolve(specifier, context) - } -}) - -const ROOT = path.resolve(import.meta.dirname, '../..') -const CHUNK_BYTES = Number(process.env.ORCA_ESCAPE_TAIL_BENCH_CHUNK_BYTES ?? '16384') -const CHUNKS = Number(process.env.ORCA_ESCAPE_TAIL_BENCH_CHUNKS ?? '640') - -for (const [name, value] of [ - ['ORCA_ESCAPE_TAIL_BENCH_CHUNK_BYTES', CHUNK_BYTES], - ['ORCA_ESCAPE_TAIL_BENCH_CHUNKS', CHUNKS] -]) { - if (!Number.isSafeInteger(value) || value <= 0) { - throw new Error(`${name} must be a positive integer, got ${value}`) - } -} - -const { advancePartialEscapeTail, extractPartialEscapeTail, MAX_PARTIAL_ESCAPE_TAIL_LENGTH } = - await import(path.join(ROOT, 'src/shared/terminal-partial-escape-tail.ts')) - -// Pre-change shape: always concatenate, always walk. function baselineAdvance(pendingTail, chunk) { const tail = extractPartialEscapeTail(pendingTail + chunk) return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail } -function buildChunk(bytes, { escapes }) { - const line = escapes - ? '\u001b[32m[build]\u001b[0m compiled src/renderer/src/components/thing.tsx in 12ms\n' - : '[build] compiled src/renderer/src/components/thing.tsx in 12ms\n' - let out = '' - while (out.length < bytes) { - out += line - } - return out.slice(0, bytes) -} +const chunkOf = (line) => line.repeat(Math.ceil(CHUNK_BYTES / line.length)).slice(0, CHUNK_BYTES) +const escFreeChunk = chunkOf('[build] compiled src/renderer/src/components/thing.tsx in 12ms\n') +const colouredChunk = chunkOf( + '\x1b[32m[build]\x1b[0m compiled src/renderer/src/components/thing.tsx in 12ms\n' +) -// Equivalence over a corpus that exercises every state the scanner can be left in, plus the -// boundaries the gate must not swallow. -const CORPUS_PIECES = [ +// Every state the scanner can be left in, plus the boundaries the gate must not swallow. +const PIECES = [ + '', 'plain output\n', - '\u001b[32mgreen\u001b[0m', - '\u001b[3', - '\u001b]0;title\u0007', - '\u001b]0;partial', - '\u001bP dcs payload', - '\u001b', - '\u0018', - '\u001a', - '\u001b]8;;https://example.com\u001b\\', - '\u001b]8;;https://example.com\u001b', - '\u001b(B', - '\u001b(', - 'tail without escapes', - '\u001b[1;2;3' + '\x1b[32mgreen\x1b[0m', + '\x1b[3', + '\x1b]0;title\x07', + '\x1b]0;partial', + '\x1bP dcs payload', + '\x1b', + '\x18', + '\x1a', + '\x1b]8;;https://example.com\x1b\\', + '\x1b]8;;https://example.com\x1b', + '\x1b(B', + '\x1b(', + '\x1b[1;2;3', + escFreeChunk ] let checked = 0 -for (const pending of CORPUS_PIECES) { - for (const chunk of CORPUS_PIECES) { - const seedTail = extractPartialEscapeTail(pending) - const expected = baselineAdvance(seedTail, chunk) - const actual = advancePartialEscapeTail(seedTail, chunk) +for (const pending of PIECES.map((piece) => extractPartialEscapeTail(piece))) { + for (const chunk of PIECES) { + const expected = baselineAdvance(pending, chunk) + const actual = advancePartialEscapeTail(pending, chunk) if (expected !== actual) { throw new Error( - `gate changed the tracked tail for pending=${JSON.stringify(seedTail)} chunk=${JSON.stringify(chunk)}: ${JSON.stringify(expected)} !== ${JSON.stringify(actual)}` + `gate changed the tracked tail: ${JSON.stringify({ pending, chunk, expected, actual })}` ) } checked += 1 } } -// A long ESC-free chunk must also agree, which is the case the gate short-circuits. -const escFreeChunk = buildChunk(CHUNK_BYTES, { escapes: false }) -if (baselineAdvance('', escFreeChunk) !== advancePartialEscapeTail('', escFreeChunk)) { - throw new Error('gate disagreed with the baseline on an ESC-free chunk') -} -checked += 1 -function median(samples) { - const sorted = [...samples].sort((left, right) => left - right) - return sorted[Math.floor(sorted.length / 2)] -} - -function timeStream(advance, chunk) { - const run = () => { +function medianMs(advance, chunk) { + // First sample is the warm-up and is discarded. + const samples = Array.from({ length: ROUNDS + 1 }, () => { + const start = performance.now() let tail = '' for (let index = 0; index < CHUNKS; index += 1) { tail = advance(tail, chunk) } - return tail - } - run() - const samples = [] - for (let round = 0; round < 7; round += 1) { - const start = performance.now() - run() - samples.push(performance.now() - start) - } - return median(samples) + return performance.now() - start + }) + return samples.slice(1).sort((left, right) => left - right)[Math.floor(ROUNDS / 2)] } -const escapedChunk = buildChunk(CHUNK_BYTES, { escapes: true }) const megabytes = ((CHUNK_BYTES * CHUNKS) / 1024 / 1024).toFixed(1) - console.log( - `Partial-escape-tail fold — ${CHUNKS} x ${CHUNK_BYTES / 1024} KB chunks (${megabytes} MB), ${checked} equivalence cases verified\n` + `Partial-escape-tail fold: ${CHUNKS} x ${CHUNK_BYTES / 1024} KB chunks (${megabytes} MB), ${checked} equivalence cases verified\n` ) console.log('| stream shape | before | after | |') console.log('| --- | --- | --- | --- |') for (const [label, chunk] of [ ['ESC-free (build logs, `cat`, piped output)', escFreeChunk], - ['SGR-coloured output (gate does not apply)', escapedChunk] + ['SGR-coloured output (gate does not apply)', colouredChunk] ]) { - const before = timeStream(baselineAdvance, chunk) - const after = timeStream(advancePartialEscapeTail, chunk) + const before = medianMs(baselineAdvance, chunk) + const after = medianMs(advancePartialEscapeTail, chunk) console.log( `| ${label} | ${before.toFixed(2)} ms | ${after.toFixed(2)} ms | ${(before / after).toFixed(1)}x |` ) diff --git a/package.json b/package.json index d1d930be708..21636f94e9c 100644 --- a/package.json +++ b/package.json @@ -144,6 +144,7 @@ "bench:agent-inspection-cadence": "node config/scripts/agent-inspection-cadence-batching-benchmark.mjs", "bench:renderer-quadratic-scans": "node config/scripts/renderer-quadratic-scan-benchmark.mjs", "bench:session-write-hot-path": "node config/scripts/session-write-hot-path-benchmark.mjs", + "bench:terminal-partial-escape-tail": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON config/scripts/terminal-partial-escape-tail-benchmark.mjs", "bench:terminal-partial-escape-tail": "node config/scripts/terminal-partial-escape-tail-benchmark.mjs", "bench:worktree-refresh-churn": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON config/scripts/worktree-refresh-churn-benchmark.mjs", "bench:multi-workspace-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-multi-workspace-typing-bench.mjs", diff --git a/src/shared/terminal-partial-escape-tail.fuzz.test.ts b/src/shared/terminal-partial-escape-tail.fuzz.test.ts new file mode 100644 index 00000000000..9f3d0b0e7b9 --- /dev/null +++ b/src/shared/terminal-partial-escape-tail.fuzz.test.ts @@ -0,0 +1,154 @@ +import { describe, expect, it } from 'vitest' +import { + advancePartialEscapeTail, + extractPartialEscapeTail, + MAX_PARTIAL_ESCAPE_TAIL_LENGTH +} from './terminal-partial-escape-tail' + +// Differential fuzz for the ESC-free gate in `advancePartialEscapeTail`: the guarded fold must be +// byte-for-byte indistinguishable from the unguarded oracle (concat + full walk + cap) on every +// input, and must preserve the fold property extract(a + b) === extract(extract(a) + b). +// A 25.6M-case out-of-band sweep (exhaustive len<=5, 2000 x 16 KB random chunks, every BMP code +// unit) found 0 divergences; this is the CI-sized slice of it. + +const oracle = (pending: string, chunk: string): string => { + const tail = extractPartialEscapeTail(pending + chunk) + return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail +} + +// Every byte class the scanner branches on, plus code units the gate's `includes` must not confuse. +const ALPHABET = [ + '\x1b', + '\x18', + '\x1a', + '\x07', + '\\', + '[', + ']', + 'P', + 'X', + '^', + '_', + '(', + '0', + ';', + 'm', + '\n', + '\x7f', + '\x9c', + 'é', + '\u{1f600}', + '\ud83d', + '\udc00' +] + +// One representative of every state the scanner can be left in. +const PENDINGS = [ + '', + '\x1b', + '\x1b[', + '\x1b[3', + '\x1b]0;ti', + '\x1b]0;ti\x1b', + '\x1bP dcs', + '\x1bPx\x1b', + '\x1b(', + '\x1b ', + '\x1b[1;2;3' +] + +const SEQUENCES = [ + '\x1b[1;31m', + '\x1b]0;my title\x07', + '\x1b]8;;https://example.com\x1b\\', + '\x1bPq#0;2;0;0;0#0!6~\x1b\\', + '\x1b(B', + '\x1b7', + '\x1b[?1049h', + '\x1b]52;c;aGVsbG8=\x1b\\', + 'ab\x1b[2Jcd' +] + +// Yields {text, depth} because an astral symbol is two UTF-16 code units: filtering on +// `text.length` would silently drop every depth-N string containing one, so the corpus would +// not be exhaustive at depth N the way the test names claim. +function* stringsUpTo(maxDepth: number): Generator<{ depth: number; text: string }> { + yield { depth: 0, text: '' } + for (let depth = 1; depth <= maxDepth; depth++) { + const digits = Array.from({ length: depth }, () => 0) + for (;;) { + yield { depth, text: digits.map((digit) => ALPHABET[digit]).join('') } + let place = depth - 1 + while (place >= 0 && ++digits[place] === ALPHABET.length) { + digits[place--] = 0 + } + if (place < 0) { + break + } + } + } +} + +describe('advancePartialEscapeTail differential fuzz', () => { + let checked = 0 + const check = (pending: string, chunk: string): void => { + checked++ + const actual = advancePartialEscapeTail(pending, chunk) + if (actual !== oracle(pending, chunk)) { + expect.fail(`gate diverged: ${JSON.stringify({ pending, chunk, actual })}`) + } + const whole = extractPartialEscapeTail(pending + chunk) + if ( + whole.length <= MAX_PARTIAL_ESCAPE_TAIL_LENGTH && + advancePartialEscapeTail(extractPartialEscapeTail(pending), chunk) !== whole + ) { + expect.fail(`fold property broke: ${JSON.stringify({ pending, chunk })}`) + } + } + + it('matches the unguarded oracle on every chunk up to length 4', () => { + for (const { text: chunk } of stringsUpTo(3)) { + for (const pending of PENDINGS) { + check(pending, chunk) + } + } + for (const { depth, text: chunk } of stringsUpTo(4)) { + if (depth === 4) { + check('', chunk) + check('\x1b[', chunk) + } + } + }) + + it('matches at every split point of known sequences', () => { + for (const sequence of SEQUENCES) { + for (let cut = 0; cut <= sequence.length; cut++) { + const afterPrefix = advancePartialEscapeTail('', sequence.slice(0, cut)) + check('', sequence.slice(0, cut)) + for (let cut2 = cut; cut2 <= sequence.length; cut2++) { + check(afterPrefix, sequence.slice(cut, cut2)) + check( + advancePartialEscapeTail(afterPrefix, sequence.slice(cut, cut2)), + sequence.slice(cut2) + ) + } + } + } + }) + + it('matches across the tail-length cap', () => { + const max = MAX_PARTIAL_ESCAPE_TAIL_LENGTH + for (const length of [max - 1, max, max + 1, max + 100]) { + const osc = `\x1b]0;${'x'.repeat(length - 4)}` + for (const chunk of ['', 'y', '\x07', '\x1b\\', '\x1b', 'plain\n', 'x'.repeat(5000)]) { + check(osc, chunk) + check('', osc + chunk) + check('\x1b]0;', osc.slice(4) + chunk) + } + } + }) + + it('ran the whole corpus', () => { + expect(checked).toBe(593_468) + }) +}) diff --git a/src/shared/terminal-partial-escape-tail.test.ts b/src/shared/terminal-partial-escape-tail.test.ts index c934abcb310..3185abc4cfc 100644 --- a/src/shared/terminal-partial-escape-tail.test.ts +++ b/src/shared/terminal-partial-escape-tail.test.ts @@ -86,46 +86,41 @@ describe('advancePartialEscapeTail', () => { }) describe('advancePartialEscapeTail ESC-free fast path', () => { - // The gate must be indistinguishable from the walk it skips: `extractPartialEscapeTail` only - // leaves `ground` on an ESC byte, so a chunk with none can only produce ''. + // Every pending-tail state the scanner can be left in x every chunk shape, asserted + // indistinguishable from the unconditional fold the gate sits in front of. const pieces = [ '', 'plain output\n', - '\u001b[32mgreen\u001b[0m', - '\u001b[3', - '\u001b]0;title\u0007', - '\u001b]0;partial', - '\u001bP dcs payload', - '\u001b', - '\u0018', - '\u001a', - '\u001b]8;;https://example.com\u001b\\', - '\u001b(', - 'no escapes at all', - '\u001b[1;2;3' + '\x1b[32mgreen\x1b[0m', + '\x1b[3', + '\x1b]0;title\x07', + '\x1b]0;partial', + '\x1bP dcs payload', + '\x1b', + '\x18', + '\x1a', + '\x1b]8;;https://example.com\x1b\\', + '\x1b(', + '\x1b[1;2;3' ] it('matches an unconditional fold for every pending-tail and chunk pairing', () => { - for (const pending of pieces) { - const seedTail = extractPartialEscapeTail(pending) + for (const pending of pieces.map((piece) => extractPartialEscapeTail(piece))) { for (const chunk of pieces) { - const unconditional = extractPartialEscapeTail(seedTail + chunk) - expect({ - pending: seedTail, - chunk, - tail: advancePartialEscapeTail(seedTail, chunk) - }).toEqual({ - pending: seedTail, - chunk, - tail: unconditional.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : unconditional - }) + // The cap belongs in the expectation: `advancePartialEscapeTail` abandons a tail over + // MAX_PARTIAL_ESCAPE_TAIL_LENGTH, so comparing it against an uncapped extract would stop + // modelling the function the moment a pairing crossed the cap. + const unguarded = extractPartialEscapeTail(pending + chunk) + expect(advancePartialEscapeTail(pending, chunk), JSON.stringify({ pending, chunk })).toBe( + unguarded.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : unguarded + ) } } }) it('still carries a pending tail through an ESC-free chunk', () => { - const pending = '\u001b]0;my-title' - const chunk = ' still inside the OSC payload' - expect(advancePartialEscapeTail(pending, chunk)).toBe(pending + chunk) + expect(advancePartialEscapeTail('\x1b]0;my-title', ' still in the OSC')).toBe( + '\x1b]0;my-title still in the OSC' + ) }) }) diff --git a/src/shared/terminal-partial-escape-tail.ts b/src/shared/terminal-partial-escape-tail.ts index 633a7264dad..b1aa44ec072 100644 --- a/src/shared/terminal-partial-escape-tail.ts +++ b/src/shared/terminal-partial-escape-tail.ts @@ -149,7 +149,7 @@ export function advancePartialEscapeTail(pendingTail: string, chunk: string): st // the full-chunk concat and the per-code-unit walk on ESC-free output (build logs, `cat`, // piped tool output) — the same gate `TerminalOscCwdTitleScanner.scan` and // `TerminalMouseModeMirror.scan` already apply on the very same ingest path. - if (pendingTail.length === 0 && !chunk.includes('\u001b')) { + if (pendingTail.length === 0 && !chunk.includes('\x1b')) { return '' } const tail = extractPartialEscapeTail(pendingTail + chunk) From cc07249e785f3d6dde3db1b44d0a0e054eff5636 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:50:37 -0700 Subject: [PATCH 24/26] fix(agent-session): refuse a pre-commit structured create with an envelope (#18697) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(agent-session): refuse a pre-commit structured create with an envelope The create route refused by throwing, which reaches a client as a generic transport error indistinguishable from a lost answer — so desktop parked the launch as visibility-unknown with no chat and no terminal. Convert the whole pre-commit span, everything before `attach`, into a refusal envelope carrying a code, and name the definitive-refusal allowlist the fallback decision needs. * fix(agent-session): gate legacy fallback on definitive refusals * fix(mobile): preserve unknown structured create outcomes --------- Co-authored-by: Merge Sim --- ...le-structured-agent-session-launch.test.ts | 72 ++++++ .../mobile-structured-agent-session-launch.ts | 18 +- ...le-session-terminal-create-actions.test.ts | 37 ++- ...ed-agent-session-precommit-refusal.test.ts | 239 ++++++++++++++++++ ...uctured-agent-session-precommit-refusal.ts | 71 ++++++ .../methods/structured-agent-session.test.ts | 37 ++- .../rpc/methods/structured-agent-session.ts | 131 ++++++---- .../launch-structured-agent-session.test.ts | 64 +++++ .../lib/launch-structured-agent-session.ts | 69 ++++- .../structured-agent-session-launch.test.ts | 26 ++ .../agent-session-definitive-refusal.test.ts | 48 ++++ .../agent-session-definitive-refusal.ts | 35 +++ 12 files changed, 764 insertions(+), 83 deletions(-) create mode 100644 src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts create mode 100644 src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts create mode 100644 src/shared/agent-session-definitive-refusal.test.ts create mode 100644 src/shared/agent-session-definitive-refusal.ts diff --git a/mobile/src/session/mobile-structured-agent-session-launch.test.ts b/mobile/src/session/mobile-structured-agent-session-launch.test.ts index 54f9b5cbe88..f575d5ac6d4 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.test.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.test.ts @@ -139,4 +139,76 @@ describe('mobile structured Codex launch', () => { kind: 'unknown' }) }) + + it.each(['structured_agent_session_unsupported', 'method_not_found'])( + 'treats a top-level %s as a definitive refusal', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'structured create unavailable' } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'failed', + message: 'structured create unavailable' + }) + } + ) + + it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])( + 'keeps a top-level %s outcome unknown', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'create outcome ambiguous' } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unknown', + message: 'create outcome ambiguous' + }) + } + ) + + it('treats an envelope unsupported refusal as definitive', async () => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { + code: 'structured_agent_session_unsupported', + message: 'structured create unavailable' + } + } + } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'failed', + message: 'structured create unavailable' + }) + }) + + it.each(['agent_session_operation_unknown', 'agent_session_ownership_unknown', 'future_code'])( + 'keeps an envelope %s refusal unknown', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { code, message: 'create outcome ambiguous' } + } + } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unknown', + message: 'create outcome ambiguous' + }) + } + ) }) diff --git a/mobile/src/session/mobile-structured-agent-session-launch.ts b/mobile/src/session/mobile-structured-agent-session-launch.ts index ecad0410dfd..b7eb8289e84 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.ts @@ -2,6 +2,7 @@ import type { AgentSessionAttachResult, AgentSessionMutationResult } from '../../../src/shared/agent-session-wire' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../src/shared/agent-session-definitive-refusal' import { structuredAgentSessionPayloadFingerprint } from '../../../src/shared/structured-agent-session-mutation' import type { RpcClient } from '../transport/rpc-client' import { structuredSessionOperationId } from './mobile-structured-agent-session-rpc' @@ -66,6 +67,13 @@ function unknownCreateResult(error: unknown): MobileStructuredCodexLaunchResult } } +function classifyCreateRefusal(code: string, message: string): MobileStructuredCodexLaunchResult { + if (!isDefinitiveAgentSessionCreateRefusal(code)) { + return unknownCreateResult(new Error(message)) + } + return { kind: 'failed', message: message || 'Could not open Codex chat.' } +} + export async function createMobileStructuredCodexSession( client: RpcClient, worktreeId: string @@ -125,10 +133,7 @@ export async function createMobileStructuredCodexSession( ) { return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) } - if (response.error.code === 'agent_session_operation_unknown') { - return unknownCreateResult(new Error(response.error.message)) - } - return { kind: 'failed', message: response.error.message || 'Could not open Codex chat.' } + return classifyCreateRefusal(response.error.code, response.error.message) } const result = response.result as AgentSessionMutationResult if (!result || typeof result !== 'object' || typeof result.ok !== 'boolean') { @@ -142,10 +147,7 @@ export async function createMobileStructuredCodexSession( ) { return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) } - if (result.refusal.code === 'agent_session_operation_unknown') { - return unknownCreateResult(new Error(result.refusal.message)) - } - return { kind: 'failed', message: result.refusal.message || 'Could not open Codex chat.' } + return classifyCreateRefusal(result.refusal.code, result.refusal.message) } if ( !result.value || diff --git a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts index c0a4e8368c5..e46bf087f38 100644 --- a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts +++ b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts @@ -137,14 +137,17 @@ describe('mobile + Codex tab creation routing', () => { expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('terminal-tab-1') }) - it('falls back to a terminal when structured creation is refused', async () => { + it('falls back to a terminal when structured creation is definitively refused', async () => { const client = clientReturning( { ok: true, result: { supported: true } }, { ok: true, result: { ok: false, - refusal: { code: 'agent_session_ownership_unknown', message: 'provider unavailable' } + refusal: { + code: 'structured_agent_session_unsupported', + message: 'provider unavailable' + } } }, terminalCreateResponse() @@ -228,4 +231,34 @@ describe('mobile + Codex tab creation routing', () => { expect(scope.setCreateError).toHaveBeenCalledWith('still unknown') expect(scope.showToast).toHaveBeenCalledWith('still unknown', 1800) }) + + it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])( + 'does not create a legacy sibling after a top-level %s response', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'create outcome ambiguous' } } + ) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + const sendRequest = client.sendRequest as unknown as ReturnType + expect(sendRequest.mock.calls.map(([method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create' + ]) + expect(scope.setCreateError).toHaveBeenCalledWith('create outcome ambiguous') + expect(scope.showToast).toHaveBeenCalledWith('create outcome ambiguous', 1800) + } + ) }) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts new file mode 100644 index 00000000000..1a62045c85b --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts @@ -0,0 +1,239 @@ +// The create route's pre-commit boundary: a failure before `attach` must reach the client as a +// refusal it can classify, and a failure at or after `attach` must not. + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' +import { setStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' +import { computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../../shared/agent-session-definitive-refusal' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { OrcaRuntimeService } from '../../orca-runtime' +import type { RpcResponse } from '../core' +import { RpcDispatcher } from '../dispatcher' +import { STRUCTURED_AGENT_SESSION_METHODS } from './structured-agent-session' + +const SESSION = 'session-alpha' +const OPERATION = '1800000000000-00000000000000000000000000000001' +const WORKTREE = 'id:workspace-1' + +const STRUCTURED_CLIENT = { + clientKind: 'runtime' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} + +function createParams(overrides: Record = {}) { + return { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree: WORKTREE, agent: 'codex' } + }), + ...(overrides.envelope as Record | undefined) + }, + worktree: WORKTREE, + agent: 'codex' + } +} + +let attach: ReturnType + +function hostStub(): StructuredAgentSessionHost { + attach = vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 0 }, + value: { sessionId: SESSION, fence: 1, page: {}, unconfirmedClientMessageIds: [] } + })) + return { attach } as unknown as StructuredAgentSessionHost +} + +const resolvedIntent = { + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'codex', + agent: 'codex', + accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + runtimeKind: 'native' +} + +async function create( + runtimeOverrides: Record = {}, + params: unknown = createParams() +): Promise { + const runtime = { + getRuntimeId: () => 'runtime-1', + registerSubscriptionCleanup: vi.fn(), + cleanupSubscription: vi.fn(), + cleanupSubscriptionsByPrefix: vi.fn(), + ensureStructuredAgentSessionHost: vi.fn(async () => undefined), + resolveStructuredAgentSessionCreateIntent: vi.fn(async (input: { envelope: unknown }) => ({ + envelope: input.envelope, + ...resolvedIntent + })), + publishStructuredAgentSessionTab: vi.fn(async () => undefined), + ...runtimeOverrides + } + const replies: RpcResponse[] = [] + await new RpcDispatcher({ + runtime: runtime as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }).dispatchStreaming( + { id: 'request-1', authToken: 'token', method: 'agentSession.create', params }, + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + STRUCTURED_CLIENT + ) + const first = replies[0] + if (!first) { + throw new Error('no reply for agentSession.create') + } + return first +} + +/** The refusal a client can act on, or null when the reply was not one. */ +function refusalOf(response: RpcResponse): { code: string; message: string } | null { + if (!response.ok) { + return null + } + const result = response.result as { ok: boolean; refusal?: { code: string; message: string } } + return result.ok ? null : (result.refusal ?? null) +} + +beforeEach(() => { + setStructuredAgentSessionHost(hostStub()) + vi.spyOn(console, 'warn').mockImplementation(() => undefined) +}) + +afterEach(() => { + setStructuredAgentSessionHost(null) + vi.restoreAllMocks() +}) + +describe('a create refused before it commits', () => { + it('answers a code-carrying refusal as a definitive envelope', async () => { + const response = await create({ + resolveStructuredAgentSessionCreateIntent: vi.fn(async () => { + throw new Error('structured_agent_session_unsupported') + }) + }) + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('structured_agent_session_unsupported') + expect(refusal?.message).toContain('structured agent chat') + expect(refusal?.message).not.toContain('Codex') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(attach).not.toHaveBeenCalled() + }) + + it('answers a code-less failure as a definitive envelope too, keeping the cause in the message', async () => { + // The class no per-site conversion catches: an unresolvable worktree throws prose, not a code. + const response = await create({ + resolveStructuredAgentSessionCreateIntent: vi.fn(async () => { + throw new Error('No worktree matches id:workspace-1') + }) + }) + + const refusal = refusalOf(response) + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(refusal?.message).toContain('structured agent chat') + expect(refusal?.message).not.toContain('Codex') + expect(refusal?.message).toContain('No worktree matches id:workspace-1') + expect(attach).not.toHaveBeenCalled() + }) + + it('answers a host that will not install as a definitive envelope', async () => { + setStructuredAgentSessionHost(null) + + const response = await create({ + ensureStructuredAgentSessionHost: vi.fn(async () => { + throw new Error('EACCES: could not open the session store') + }) + }) + + const refusal = refusalOf(response) + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(refusal?.message).toContain('could not open the session store') + }) + + it('answers a missing host as a definitive envelope rather than a thrown code', async () => { + setStructuredAgentSessionHost(null) + + const response = await create() + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('structured_agent_session_unsupported') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + }) + + it('still refuses a fingerprint conflict with its own code, not a pre-commit one', async () => { + const response = await create( + {}, + createParams({ envelope: { payloadFingerprint: 'a'.repeat(64) } }) + ) + + expect(refusalOf(response)?.code).toBe('agent_session_operation_conflict') + expect(attach).not.toHaveBeenCalled() + }) +}) + +describe('the boundary the envelope stops at', () => { + it('leaves a failure at attach unknown, because it may have committed', async () => { + attach.mockRejectedValueOnce(new Error('attach exploded')) + + const response = await create() + + expect(response).toMatchObject({ ok: false, error: { code: 'runtime_error' } }) + expect(refusalOf(response)).toBeNull() + }) + + it('leaves a committed create whose tab could not be published unknown', async () => { + const response = await create({ + publishStructuredAgentSessionTab: vi.fn(async () => { + throw new Error('publish failed') + }) + }) + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('agent_session_operation_unknown') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(false) + }) + + it('keeps hiding the surface from a client that never advertised it', async () => { + const replies: RpcResponse[] = [] + await new RpcDispatcher({ + runtime: { getRuntimeId: () => 'runtime-1' } as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }).dispatchStreaming( + { + id: 'request-1', + authToken: 'token', + method: 'agentSession.create', + params: createParams() + }, + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + { clientKind: 'runtime', clientCapabilities: [] } + ) + + expect(replies[0]).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + }) + + it('keeps a client-declared fence a programming error, not a refusal', async () => { + const response = await create({}, createParams({ envelope: { expectedRuntimeFence: 1 } })) + + expect(response).toMatchObject({ + ok: false, + error: { code: 'agent_session_operation_invalid' } + }) + }) +}) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts new file mode 100644 index 00000000000..24d9fa2aec4 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts @@ -0,0 +1,71 @@ +// Nothing before `attach` commits a session, so every failure in that span definitively created +// nothing. Thrown, it reaches a remote client as a generic transport error, indistinguishable from +// an answer that was lost on the way back — and a client that cannot tell those apart either +// strands the user with no chat and no terminal, or spawns a sibling beside a session that may +// already exist. So the whole span answers with a refusal envelope carrying a code, whatever it +// failed on. +// +// Converting the span rather than each throw site is deliberate: alongside the throws that carry a +// code there is a code-less class — an unresolvable worktree, a store that will not open, a host +// that will not install — that no per-site list catches, and it is exactly the class that reaches +// the user as nothing at all. + +import { + AGENT_SESSION_WIRE_REFUSAL_CODES, + type AgentSessionWireRefusal, + type AgentSessionWireRefusalCode +} from '../../../../shared/agent-session-wire' + +export type StructuredCreateRefused = { refusal: AgentSessionWireRefusal } + +/** A pre-commit failure with no code of its own still proves the host could not serve a structured + * session for this request and did not create one, which is what `unsupported` says on the wire. + * A new code would say it more precisely, but only to clients new enough to know it. */ +const UNCODED_PRECOMMIT_REFUSAL_CODE: AgentSessionWireRefusalCode = + 'structured_agent_session_unsupported' + +function wireRefusalCode(error: unknown): AgentSessionWireRefusalCode | null { + const candidates = [ + error instanceof Error && 'code' in error ? (error as { code: unknown }).code : undefined, + error instanceof Error ? error.message : String(error) + ] + for (const candidate of candidates) { + if ( + typeof candidate === 'string' && + (AGENT_SESSION_WIRE_REFUSAL_CODES as readonly string[]).includes(candidate) + ) { + return candidate as AgentSessionWireRefusalCode + } + } + return null +} + +function precommitRefusal(error: unknown): AgentSessionWireRefusal { + const code = wireRefusalCode(error) + if (code) { + return { code, message: 'Orca cannot open a structured agent chat for this workspace.' } + } + const message = error instanceof Error ? error.message : String(error) + // A code-less failure here is often a defect, not a policy answer; the refusal keeps the user + // moving, the log keeps the cause findable. + console.warn('[agent-session] create refused before it committed anything', error) + return { + code: UNCODED_PRECOMMIT_REFUSAL_CODE, + message: `Orca could not prepare a structured agent chat for this workspace: ${message}` + } +} + +/** + * Runs the pre-commit half of a create. Anything it throws becomes a refusal; a refusal it returns + * itself passes through. Must not wrap `attach` or anything after it — past that point a failure no + * longer proves the session does not exist. + */ +export async function resolveUncommittedStructuredCreate( + prepare: () => Promise +): Promise { + try { + return await prepare() + } catch (error) { + return { refusal: precommitRefusal(error) } + } +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index 155d0aa6768..e4888e4a9df 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -436,21 +436,34 @@ describe('method routing', () => { /** A client-supplied location skips the worktree-resolving support check, so both attach-shaped * entries must ask the executing host directly or a host that cannot fence a provider child * would create one anyway. */ - it.each(['agentSession.create', 'agentSession.ensure'])( - 'refuses %s for a client-supplied location the executing host does not support', - async (method) => { - hostCalls.supportsCreate.mockReturnValue(false) + it('returns a refusal envelope when create cannot support a client-supplied location', async () => { + hostCalls.supportsCreate.mockReturnValue(false) - const refused = await call(method, attachParams()) + const refused = await call('agentSession.create', attachParams()) - expect(refused).toMatchObject({ + expect(refused).toMatchObject({ + ok: true, + result: { ok: false, - error: { message: expect.stringContaining('structured_agent_session_unsupported') } - }) - expect(hostCalls.attach).not.toHaveBeenCalled() - expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') - } - ) + refusal: { code: 'structured_agent_session_unsupported' } + } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + }) + + it('keeps ensure failures as top-level errors for an unsupported client location', async () => { + hostCalls.supportsCreate.mockReturnValue(false) + + const refused = await call('agentSession.ensure', attachParams()) + + expect(refused).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + }) it('tags the prompt kind from the method name, not from the client', async () => { const params = { diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index abe196bd636..b69ff6fd628 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -20,6 +20,7 @@ import { } from './structured-agent-session-gate' import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold' +import { resolveUncommittedStructuredCreate } from './structured-agent-session-precommit-refusal' import { AttachParams, CancelParams, @@ -51,21 +52,27 @@ function subscriptionIdFor(ctx: RpcContext, sessionId: string): string { * host the same question directly: the answer includes host-measured facts the client cannot see * or forge, such as whether this machine can read a provider child's process start time. */ -async function attachClientSuppliedLocation( - params: z.infer, - ctx: RpcContext -): Promise { +async function resolveClientSuppliedAttach(params: z.infer, ctx: RpcContext) { await ensureHostInstalled(ctx) const host = requireHost(ctx) if (!host.supportsCreate(params.location, params.agent)) { throw new Error('structured_agent_session_unsupported') } const { agent: _attachAgent, provider: _attachProvider, ...attachWithoutAgent } = params - return host.attach(callerFor(ctx), { + const attachParams = { ...attachWithoutAgent, provider: params.provider as 'claude' | 'codex', agent: params.agent as 'claude' | 'codex' - } as AgentSessionAttachParams) + } as AgentSessionAttachParams + return { host, attachParams } +} + +async function attachClientSuppliedLocation( + params: z.infer, + ctx: RpcContext +): Promise { + const { host, attachParams } = await resolveClientSuppliedAttach(params, ctx) + return host.attach(callerFor(ctx), attachParams) } export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ @@ -87,60 +94,76 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ if (params.envelope.expectedRuntimeFence !== null) { throw new Error('agent_session_operation_invalid') } - if ('worktree' in params) { - const intentFingerprint = computeAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: params.envelope.sessionId, - fields: { worktree: params.worktree, agent: params.agent } - }) - const conflict = agentSessionFingerprintConflict(params.envelope, intentFingerprint) - if (conflict) { - return { ok: false, refusal: conflict } - } - const resolved = await ctx.runtime.resolveStructuredAgentSessionCreateIntent(params) - const hostFingerprint = computeAgentSessionPayloadFingerprint({ - method: 'agentSession.attach', - sessionId: params.envelope.sessionId, - fields: { - location: resolved.location, - provider: resolved.provider, - agent: resolved.agent, - accountHome: resolved.accountHome, - runtimeKind: resolved.runtimeKind, - expectedRuntimeFence: null + // Everything up to `attach` is pre-commit, and answers with a refusal rather than a throw so + // a client can tell "nothing was created" from "the outcome is unknown". + const prepared = await resolveUncommittedStructuredCreate(async () => { + if ('worktree' in params) { + const intentFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: params.envelope.sessionId, + fields: { worktree: params.worktree, agent: params.agent } + }) + const conflict = agentSessionFingerprintConflict(params.envelope, intentFingerprint) + if (conflict) { + return { refusal: conflict } } - }) - await ensureHostInstalled(ctx) - const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved - const attachParams: AgentSessionAttachParams = { - ...resolvedAttach, - provider: resolved.provider as 'claude' | 'codex', - agent: resolved.agent as 'claude' | 'codex', - envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } - } - const result = await requireHost(ctx).attach(callerFor(ctx), attachParams) - if (result.ok) { - try { - await ctx.runtime.publishStructuredAgentSessionTab({ + const resolved = await ctx.runtime.resolveStructuredAgentSessionCreateIntent(params) + const hostFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.attach', + sessionId: params.envelope.sessionId, + fields: { + location: resolved.location, + provider: resolved.provider, + agent: resolved.agent, + accountHome: resolved.accountHome, + runtimeKind: resolved.runtimeKind, + expectedRuntimeFence: null + } + }) + await ensureHostInstalled(ctx) + const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved + const attachParams: AgentSessionAttachParams = { + ...resolvedAttach, + provider: resolved.provider as 'claude' | 'codex', + agent: resolved.agent as 'claude' | 'codex', + envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } + } + return { + host: requireHost(ctx), + attachParams, + tab: { workspaceId: resolved.location.workspaceId, - sessionId: result.value.sessionId, - agent: resolved.agent as 'claude' | 'codex', - activate: true - }) - } catch (error) { - console.warn('[agent-session] create committed before tab publication failed', error) - return { - ok: false, - refusal: { - code: 'agent_session_operation_unknown', - message: 'The chat may have been created, but its tab could not be confirmed.' - } + agent: resolved.agent as 'claude' | 'codex' } } } - return result + const { host, attachParams } = await resolveClientSuppliedAttach(params, ctx) + return { host, attachParams, tab: null } + }) + if ('refusal' in prepared) { + return { ok: false, refusal: prepared.refusal } } - return attachClientSuppliedLocation(params, ctx) + const result = await prepared.host.attach(callerFor(ctx), prepared.attachParams) + if (result.ok && prepared.tab) { + try { + await ctx.runtime.publishStructuredAgentSessionTab({ + workspaceId: prepared.tab.workspaceId, + sessionId: result.value.sessionId, + agent: prepared.tab.agent, + activate: true + }) + } catch (error) { + console.warn('[agent-session] create committed before tab publication failed', error) + return { + ok: false, + refusal: { + code: 'agent_session_operation_unknown', + message: 'The chat may have been created, but its tab could not be confirmed.' + } + } + } + } + return result } }), defineMethod({ diff --git a/src/renderer/src/lib/launch-structured-agent-session.test.ts b/src/renderer/src/lib/launch-structured-agent-session.test.ts index 5f46d5b60d2..e9f65f3477b 100644 --- a/src/renderer/src/lib/launch-structured-agent-session.test.ts +++ b/src/renderer/src/lib/launch-structured-agent-session.test.ts @@ -3,6 +3,7 @@ import { structuredAgentSessionPayloadFingerprint } from '../../../shared/struct import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' import { createStructuredAgentSessionLaunchIntent, + isDefinitiveStructuredAgentSessionCreateError, launchStructuredAgentSession, StructuredAgentSessionCreateRefusalError } from './launch-structured-agent-session' @@ -241,4 +242,67 @@ describe('structured agent session launch', () => { expect(second).toBe(first) expect(intent.params.envelope.clientOperationId).toMatch(/^\d{13}-[0-9a-f]{32}$/) }) + + it('preserves an unknown refusal code without classifying it as fallback-safe', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: false, + refusal: { + code: 'agent_session_operation_unknown', + message: 'The chat may already exist.' + } + }) + + const error = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-unknown', 'codex') + ).catch((caught: unknown) => caught) + + expect(error).not.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(error).toMatchObject({ code: 'agent_session_operation_unknown' }) + expect(isDefinitiveStructuredAgentSessionCreateError(error)).toBe(false) + }) + + it('preserves a definitive refusal code for the fallback path', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: false, + refusal: { + code: 'structured_agent_session_unsupported', + message: 'Structured chat is unavailable.' + } + }) + + const error = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-unsupported', 'codex') + ).catch((caught: unknown) => caught) + + expect(error).toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(error).toMatchObject({ code: 'structured_agent_session_unsupported' }) + expect(isDefinitiveStructuredAgentSessionCreateError(error)).toBe(true) + }) + + it.each(['method_not_found', 'structured_agent_session_unsupported'])( + 'turns an old-host %s error into a definitive transport refusal', + async (code) => { + vi.mocked(callStructuredAgentSession).mockRejectedValueOnce( + Object.assign(new Error(code), { code }) + ) + const oldHostError = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent(`workspace-old-host-${code}`, 'codex') + ).catch((caught: unknown) => caught) + + expect(oldHostError).toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(oldHostError).toMatchObject({ code }) + } + ) + + it('keeps an unclassified transport failure outcome unknown', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValueOnce( + Object.assign(new Error('Connection lost'), { code: 'runtime_error' }) + ) + const transportError = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-offline', 'codex') + ).catch((caught: unknown) => caught) + + expect(transportError).not.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(isDefinitiveStructuredAgentSessionCreateError(transportError)).toBe(false) + }) }) diff --git a/src/renderer/src/lib/launch-structured-agent-session.ts b/src/renderer/src/lib/launch-structured-agent-session.ts index 0d12ca91d40..6c2d1694437 100644 --- a/src/renderer/src/lib/launch-structured-agent-session.ts +++ b/src/renderer/src/lib/launch-structured-agent-session.ts @@ -9,6 +9,7 @@ import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' import { hasRuntimeRpcErrorCode } from '../../../shared/runtime-rpc-error-code' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../shared/agent-session-definitive-refusal' import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' import { useAppStore } from '@/store' @@ -32,7 +33,36 @@ export type StructuredAgentSessionLaunchIntent = { params: StructuredAgentSessionCreateParams } -export class StructuredAgentSessionCreateRefusalError extends Error {} +export class StructuredAgentSessionCreateRefusalError extends Error { + constructor( + message: string, + readonly code: string = 'structured_agent_session_unsupported' + ) { + super(message) + this.name = 'StructuredAgentSessionCreateRefusalError' + } +} + +const DEFINITIVE_CREATE_FAILURE_CODES = [ + 'structured_agent_session_unsupported', + 'method_not_found' +] as const + +function definitiveStructuredAgentSessionCreateErrorCode(error: unknown): string | null { + if (error instanceof StructuredAgentSessionCreateRefusalError) { + return isDefinitiveAgentSessionCreateRefusal(error.code) ? error.code : null + } + for (const code of DEFINITIVE_CREATE_FAILURE_CODES) { + if (hasRuntimeRpcErrorCode(error, code)) { + return code + } + } + return null +} + +export function isDefinitiveStructuredAgentSessionCreateError(error: unknown): boolean { + return definitiveStructuredAgentSessionCreateErrorCode(error) !== null +} export function createStructuredAgentSessionLaunchIntent( worktreeId: string, @@ -140,7 +170,10 @@ async function requireHostCreateSupport(intent: StructuredAgentSessionLaunchInte } if (!(await hostSupportsCreate(intent))) { abandonStructuredAgentSessionLaunchIntent(intent) - throw new StructuredAgentSessionCreateRefusalError('structured_agent_session_unsupported') + throw new StructuredAgentSessionCreateRefusalError( + 'structured_agent_session_unsupported', + 'structured_agent_session_unsupported' + ) } } @@ -148,12 +181,34 @@ export async function launchStructuredAgentSession( intent: StructuredAgentSessionLaunchIntent ): Promise> { await requireHostCreateSupport(intent) - const result = await callStructuredAgentSession< - AgentSessionMutationResult - >({ kind: 'local' }, 'agentSession.create', intent.params) + let result: AgentSessionMutationResult + try { + result = await callStructuredAgentSession>( + { kind: 'local' }, + 'agentSession.create', + intent.params + ) + } catch (error) { + const code = definitiveStructuredAgentSessionCreateErrorCode(error) + if (code) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw new StructuredAgentSessionCreateRefusalError( + error instanceof Error ? error.message : String(error), + code + ) + } + throw error + } if (!result.ok) { - abandonStructuredAgentSessionLaunchIntent(intent) - throw new StructuredAgentSessionCreateRefusalError(result.refusal.message) + const error = new StructuredAgentSessionCreateRefusalError( + result.refusal.message, + result.refusal.code + ) + if (isDefinitiveStructuredAgentSessionCreateError(error)) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw error + } + throw Object.assign(new Error(error.message), { code: error.code }) } return { sessionId: result.value.sessionId, fence: result.value.fence } } diff --git a/src/renderer/src/lib/structured-agent-session-launch.test.ts b/src/renderer/src/lib/structured-agent-session-launch.test.ts index f86f2432b99..0363cfb7566 100644 --- a/src/renderer/src/lib/structured-agent-session-launch.test.ts +++ b/src/renderer/src/lib/structured-agent-session-launch.test.ts @@ -476,6 +476,32 @@ describe('startStructuredAgentLaunch', () => { expect(retryFallback).toHaveBeenCalledOnce() }) + it('never starts a sibling fallback for a post-attach unknown refusal', async () => { + const worktreeId = 'wt-post-attach-unknown' + const intent = launchIntent(worktreeId) + const fallback = vi.fn() + mocks.createIntent.mockReturnValueOnce(intent) + mocks.launch.mockRejectedValue( + Object.assign(new Error('The chat may already exist.'), { + code: 'agent_session_operation_unknown' + }) + ) + vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) + + const launch = startStructuredAgentLaunch(worktreeId, 'codex') + const fallbackResult = launch.claimDefinitiveRefusalFallback(fallback) + + await expect(launch.launchResult).rejects.toMatchObject({ + code: 'agent_session_operation_unknown' + }) + expect(launch.isVisibilityUnknown()).toBe(true) + expect(launch.releaseCallerAfterUnknownOutcome()).toBe(true) + await expect(fallbackResult).resolves.toBe(false) + expect(fallback).not.toHaveBeenCalled() + expect(mocks.createIntent).toHaveBeenCalledOnce() + expect(mocks.launch).toHaveBeenCalledTimes(2) + }) + it('releases a definitively refused intent so a new click can create a new identity', async () => { const worktreeId = 'wt-refused' const first = launchIntent(worktreeId, 'session-first') diff --git a/src/shared/agent-session-definitive-refusal.test.ts b/src/shared/agent-session-definitive-refusal.test.ts new file mode 100644 index 00000000000..74dd3529cf6 --- /dev/null +++ b/src/shared/agent-session-definitive-refusal.test.ts @@ -0,0 +1,48 @@ +import { describe, expect, it } from 'vitest' +import { AGENT_SESSION_WIRE_REFUSAL_CODES } from './agent-session-wire' +import { agentSessionRefusalOperationState } from './agent-session-refusal-retry' +import { isDefinitiveAgentSessionCreateRefusal } from './agent-session-definitive-refusal' + +describe('definitive agent-session create refusals', () => { + it('treats an unsupported structured session as definitive', () => { + expect(isDefinitiveAgentSessionCreateRefusal('structured_agent_session_unsupported')).toBe(true) + }) + + it('never treats an unproven outcome as definitive', () => { + expect(isDefinitiveAgentSessionCreateRefusal('agent_session_operation_unknown')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('agent_session_ownership_unknown')).toBe(false) + }) + + it('leaves transport failures, timeouts and a missing code unknown', () => { + expect(isDefinitiveAgentSessionCreateRefusal('runtime_error')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('remote_runtime_unavailable')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('timeout')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('runtime_timeout')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal(undefined)).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal(null)).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('')).toBe(false) + }) + + it('counts a method an old host never registered as definitive', () => { + expect(isDefinitiveAgentSessionCreateRefusal('method_not_found')).toBe(true) + }) + + it('is an allowlist: every other wire refusal code is unknown', () => { + const definitive = AGENT_SESSION_WIRE_REFUSAL_CODES.filter((code) => + isDefinitiveAgentSessionCreateRefusal(code) + ) + expect(definitive).toEqual(['structured_agent_session_unsupported']) + }) + + it('does not answer the durable-settlement question, which disagrees on the one code that matters', () => { + // Guards the reuse this allowlist exists to avoid: settlement state calls the definitive + // refusal pending-admission, which would rule out the fallback it is meant to allow. + expect( + agentSessionRefusalOperationState( + 'agentSession.create', + 'structured_agent_session_unsupported' + ) + ).toBe('pending-admission') + expect(isDefinitiveAgentSessionCreateRefusal('structured_agent_session_unsupported')).toBe(true) + }) +}) diff --git a/src/shared/agent-session-definitive-refusal.ts b/src/shared/agent-session-definitive-refusal.ts new file mode 100644 index 00000000000..80721592ed9 --- /dev/null +++ b/src/shared/agent-session-definitive-refusal.ts @@ -0,0 +1,35 @@ +/** + * "May a caller create something else instead?" — the fallback question. + * + * Deliberately NOT `agentSessionRefusalOperationState`: that answers "did this operation durably + * settle?", and for that question `structured_agent_session_unsupported` is correctly + * pending-admission. Reused here it would rule out a fallback on the one refusal that most needs + * one. The two questions only look alike. + * + * An allowlist, never a negation: falling back on an outcome the host could not describe is how a + * user ends up with two sessions for one intent. Everything absent — transport failures, timeouts, + * `agent_session_operation_unknown`, `agent_session_ownership_unknown` — is unknown, and unknown + * never falls back. + */ + +import type { AgentSessionWireRefusalCode } from './agent-session-wire' + +/** Proves the host neither created a session nor will on a retry. */ +const DEFINITIVE_REFUSAL_CODES: ReadonlySet = new Set([ + 'structured_agent_session_unsupported' +]) + +/** A dispatcher that never registered the method ran no handler at all, which is as definitive as + * a refusal — and the only transport-level answer that is. */ +const DEFINITIVE_RPC_ERROR_CODES: ReadonlySet = new Set(['method_not_found']) + +/** + * True only when the code proves nothing was created. Accepts a wire refusal code or an RPC error + * code; the two namespaces are disjoint. + */ +export function isDefinitiveAgentSessionCreateRefusal(code: string | null | undefined): boolean { + if (typeof code !== 'string') { + return false + } + return DEFINITIVE_REFUSAL_CODES.has(code) || DEFINITIVE_RPC_ERROR_CODES.has(code) +} From ccf3e27800b9a9e820f26723677d9a88fcbda07a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:56:50 -0700 Subject: [PATCH 25/26] perf(relay): bound the symlink directory probes a remote readDir fans out (#18752) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `readRelayDir` issued one `stat` per symlinked entry and awaited them all in a single `Promise.all`. A pnpm `node_modules` is hundreds-to-thousands of package symlinks in one directory, so expanding it over SSH put that many stats in flight at once, saturating libuv's four-thread pool and delaying every other relay filesystem operation — including the interactive reads `fs-list-files-scan-coordinator` exists to protect. The probes now run through `forEachWithConcurrency` at 8, the cap every other bounded probe in this codebase already uses (`GIT_COMMON_SNAPSHOT_CONCURRENCY`, `PRUNABLE_EXISTENCE_PROBE_CONCURRENCY`, `SPARSE_CHECKOUT_DETECTION_CONCURRENCY`). Results and ordering are unchanged: every symlink still resolves to its target's kind, and `sortDirEntries` still runs afterwards. The new test builds a 60-symlink directory and asserts the same 60 stats happen with exactly 8 in flight at peak — the probes overlap, and never past the cap. --- src/relay/fs-path-metadata-requests.ts | 27 ++++--- ...-path-metadata-symlink-concurrency.test.ts | 70 +++++++++++++++++++ 2 files changed, 89 insertions(+), 8 deletions(-) create mode 100644 src/relay/fs-path-metadata-symlink-concurrency.test.ts diff --git a/src/relay/fs-path-metadata-requests.ts b/src/relay/fs-path-metadata-requests.ts index 2a9717a6484..b0fa2347d95 100644 --- a/src/relay/fs-path-metadata-requests.ts +++ b/src/relay/fs-path-metadata-requests.ts @@ -1,6 +1,8 @@ import { readdir, stat, lstat, realpath } from 'node:fs/promises' +import type { Dirent } from 'node:fs' import { join } from 'node:path' import { sortDirEntries } from '../shared/file-name-sort' +import { forEachWithConcurrency } from '../shared/map-with-concurrency' import { expandTilde } from './context' async function resolveSymlinkDirectoryEntry( @@ -34,11 +36,18 @@ function fileStatFromLstat(stats: Awaited>) { } } +// Why bounded: a pnpm `node_modules` is hundreds-to-thousands of package symlinks, and one +// unbounded `Promise.all` of stats from a single readDir saturates libuv's four-thread pool — +// delaying every other relay filesystem operation, including the interactive reads the +// list-files scan coordinator exists to protect. Matches the cap every other bounded probe in +// this codebase uses. +const SYMLINK_DIRECTORY_PROBE_CONCURRENCY = 8 + export async function readRelayDir(params: Record) { const dirPath = expandTilde(params.dirPath as string) const entries = await readdir(dirPath, { withFileTypes: true }) const mapped: { name: string; isDirectory: boolean; isSymlink: boolean }[] = [] - const symlinkProbes: Promise[] = [] + const symlinkEntries: { entry: Dirent; mappedEntry: (typeof mapped)[number] }[] = [] for (const entry of entries) { const mappedEntry = { name: entry.name, @@ -47,15 +56,17 @@ export async function readRelayDir(params: Record) { } mapped.push(mappedEntry) if (!mappedEntry.isDirectory && mappedEntry.isSymlink) { - symlinkProbes.push( - resolveSymlinkDirectoryEntry(dirPath, entry).then((isDirectory) => { - mappedEntry.isDirectory = isDirectory - }) - ) + symlinkEntries.push({ entry, mappedEntry }) } } - if (symlinkProbes.length > 0) { - await Promise.all(symlinkProbes) + if (symlinkEntries.length > 0) { + await forEachWithConcurrency( + symlinkEntries, + SYMLINK_DIRECTORY_PROBE_CONCURRENCY, + async ({ entry, mappedEntry }) => { + mappedEntry.isDirectory = await resolveSymlinkDirectoryEntry(dirPath, entry) + } + ) } return sortDirEntries(mapped) } diff --git a/src/relay/fs-path-metadata-symlink-concurrency.test.ts b/src/relay/fs-path-metadata-symlink-concurrency.test.ts new file mode 100644 index 00000000000..3a342a79e7c --- /dev/null +++ b/src/relay/fs-path-metadata-symlink-concurrency.test.ts @@ -0,0 +1,70 @@ +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as FsPromisesModule from 'node:fs/promises' + +const statCalls = vi.hoisted(() => ({ inFlight: 0, peak: 0, total: 0 })) + +vi.mock('node:fs/promises', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + stat: async (...args: Parameters) => { + statCalls.inFlight += 1 + statCalls.total += 1 + statCalls.peak = Math.max(statCalls.peak, statCalls.inFlight) + try { + return await actual.stat(...args) + } finally { + statCalls.inFlight -= 1 + } + } + } +}) + +const { readRelayDir } = await import('./fs-path-metadata-requests') + +describe('relay readDir symlink probes', () => { + let root: string + let targetRoot: string + + beforeEach(() => { + statCalls.inFlight = 0 + statCalls.peak = 0 + statCalls.total = 0 + root = mkdtempSync(join(tmpdir(), 'orca-relay-readdir-')) + // Kept outside `root` so the listing contains only the symlinks under test. + targetRoot = mkdtempSync(join(tmpdir(), 'orca-relay-readdir-target-')) + const target = join(targetRoot, 'target') + mkdirSync(target) + writeFileSync(join(target, 'index.js'), '') + // A pnpm-shaped node_modules: many package symlinks in one directory. Junctions on + // Windows: plain symlinks need Developer Mode there. + for (let index = 0; index < 60; index += 1) { + symlinkSync( + target, + join(root, `pkg-${index}`), + process.platform === 'win32' ? 'junction' : 'dir' + ) + } + }) + + afterEach(() => { + rmSync(root, { recursive: true, force: true }) + rmSync(targetRoot, { recursive: true, force: true }) + }) + + it('bounds concurrent symlink stats instead of issuing one per entry at once', async () => { + const entries = await readRelayDir({ dirPath: root }) + + expect(statCalls.total).toBe(60) + // Exactly the cap: every worker enters `stat` before any resolves, so the peak proves the + // probes overlap and that no more than 8 ever do. Unbounded, all 60 would be in flight, + // saturating libuv's four-thread pool and stalling every other relay filesystem read. + expect(statCalls.peak).toBe(8) + // Behaviour is unchanged: every symlink still resolves to its target's kind. + expect(entries).toHaveLength(60) + expect(entries.every((entry) => entry.isDirectory && entry.isSymlink)).toBe(true) + }) +}) From e95d247be1e0cd5e0ff1841931d4fe23a5c0f92b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:57:02 -0700 Subject: [PATCH 26/26] perf(terminal): cheap-tier process inspection for anchored local agent panes (#18780) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(terminal): cheap-tier process inspection for anchored local agent panes Every idle local pane's completion cadence ran a full whole-host `ps` (with `tty=` and `command=`, 0.34-0.50s on a 1,900-process Mac, 1.15s on Linux) purely to build `foregroundProcessEvidence` that the renderer then discards for local ids. Add a cheap tier (same job-control columns, no tty/command, 0.03s) gated so that it introduces no user-facing trade-off: - Only a pane whose last FULL capture proved a recognized agent may take the cheap tier. Panes with no anchor always take the full capture, so start discovery keeps today's exact behaviour. - The cheap tick compares a per-pane fingerprint (root shell pid+start, tpgid, every descendant's pid+start+pgid+job-control state). Any change, a changed node-pty foreground name, an unreadable capture, or an incarnation mismatch escalates to the full capture. A recognized agent's exit is always a pid vanishing, which the fingerprint always sees. - A cheap answer OMITS evidence rather than fabricating a tty-less fence. Remote/restore consumers never send `steadyState`, so they keep the full capture unchanged. - `steadyState` is a new optional request field; an old daemon ignores it and answers with the full capture. Measured (8 idle panes, 60s, idle cadence, forks counted by column set): 30 full -> 1 full + 29 cheap. * fix(terminal): route the cheap ps capture through runProcess The cheap-tier reader imported node:child_process directly, which the child-process import-boundary and windowsHide ratchet tests reject (CI shards 1/8 and 3/8). Use Orca's single spawn entry point instead; it pins windowsHide and encodes argv. Map its result onto the capture-error vocabulary: outputTruncated -> capture_truncated, timedOut -> capture_timeout, non-zero exit -> ps_exit_. Tests mock at the runProcess seam. * fix(perf): refuse a pane fingerprint when any descendant start marker is missing `buildPaneProcessFingerprint` rejected only a missing root start marker; a missing descendant marker was stamped as `?`. Two captures that both failed to read the same descendant therefore compared equal, which removes the pid-reuse protection the fingerprint exists to provide: a recycled pid could make a vanished agent look unchanged, and the cheap tier would keep serving its name instead of escalating. Reachable on Linux, where `readLinuxProcStartTime` legitimately returns null when a process exits between the `ps` capture and the `/proc//stat` read. Every subtree member now needs a start marker or the fingerprint is refused, which sends the caller to the full capture — the same conservative default every other uncertain path takes. Reported by CodeRabbit on #18780. The two new tests fail against the previous code with `expected '4242@2400#4300:|4300@?:4300:+' to be null`. --- .../daemon-foreground-process-protocol.ts | 2 + ...on-pty-adapter-steady-state-compat.test.ts | 78 +++++ .../daemon/daemon-pty-process-inspection.ts | 6 +- src/main/daemon/daemon-pty-router.ts | 2 +- src/main/daemon/daemon-request-router.ts | 15 +- .../foreground-process-tracker.ts | 9 +- src/main/daemon/session-subprocess-handle.ts | 2 +- src/main/daemon/session.ts | 4 +- ...nal-host-cheap-tier-ps-scan-volume.test.ts | 217 +++++++++++++ ...host-process-inspection-cheap-tier.test.ts | 297 ++++++++++++++++++ .../terminal-host-process-inspection.ts | 60 +++- .../terminal-host-steady-state-anchor.ts | 58 ++++ src/main/daemon/terminal-host.ts | 3 +- src/main/ipc/pty/ipc/inspect.ts | 10 +- ...y-foreground-inspection-cheap-tier.test.ts | 138 ++++++++ .../local-pty-foreground-inspection.ts | 63 +++- .../providers/local-pty-provider-state.ts | 9 +- .../posix-pane-foreground-fingerprint.test.ts | 187 +++++++++++ .../posix-pane-foreground-fingerprint.ts | 93 ++++++ src/main/providers/pty-process-inspection.ts | 3 + src/preload/api/pty-api.ts | 6 +- .../pty-bridge-stream-and-serialization.ts | 6 +- .../agent-completion-coordinator-types.ts | 2 +- .../agent-completion-process-monitor.ts | 16 +- ...ent-completion-steady-state-opt-in.test.ts | 84 +++++ .../runtime/runtime-terminal-inspection.ts | 2 +- .../cheap-process-table-snapshot-reader.ts | 51 +++ .../cheap-process-table-snapshot.test.ts | 147 +++++++++ src/shared/process-table-snapshot-reader.ts | 28 +- src/shared/process-table-snapshot.ts | 54 ++++ 30 files changed, 1614 insertions(+), 38 deletions(-) create mode 100644 src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts create mode 100644 src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts create mode 100644 src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts create mode 100644 src/main/daemon/terminal-host-steady-state-anchor.ts create mode 100644 src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts create mode 100644 src/main/providers/posix-pane-foreground-fingerprint.test.ts create mode 100644 src/main/providers/posix-pane-foreground-fingerprint.ts create mode 100644 src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts create mode 100644 src/shared/cheap-process-table-snapshot-reader.ts create mode 100644 src/shared/cheap-process-table-snapshot.test.ts diff --git a/src/main/daemon/daemon-foreground-process-protocol.ts b/src/main/daemon/daemon-foreground-process-protocol.ts index 25c6113057c..5c29e8a9e31 100644 --- a/src/main/daemon/daemon-foreground-process-protocol.ts +++ b/src/main/daemon/daemon-foreground-process-protocol.ts @@ -18,5 +18,7 @@ export type InspectProcessRequest = Omit & type: 'inspectProcess' payload: GetForegroundProcessRequest['payload'] & { expectedIncarnationId?: string + /** Optional; a daemon that predates it answers with the full capture as it always did. */ + steadyState?: boolean } } diff --git a/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts b/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts new file mode 100644 index 00000000000..ae7c29847c9 --- /dev/null +++ b/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts @@ -0,0 +1,78 @@ +import { describe, expect, it, vi } from 'vitest' +import { DaemonPtyAdapter } from './daemon-pty-adapter' +import { COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION, PROTOCOL_VERSION } from './types' + +type ClientInternals = { + client: { request: ReturnType; disconnect: ReturnType } +} + +function createAdapter( + protocolVersion: number, + request: ReturnType +): DaemonPtyAdapter { + const adapter = new DaemonPtyAdapter({ + socketPath: '/tmp/orca-steady-state-compat.sock', + tokenPath: '/tmp/orca-steady-state-compat.token', + protocolVersion + }) + ;(adapter as unknown as ClientInternals).client = { request, disconnect: vi.fn() } + return adapter +} + +describe('steadyState across daemon versions', () => { + it('sends steadyState as an additive optional field on the existing inspectProcess request', async () => { + const request = vi.fn(async () => ({ foregroundProcess: 'claude', hasChildProcesses: true })) + const adapter = createAdapter(PROTOCOL_VERSION, request) + await adapter.inspectProcess('sess-a', { steadyState: true }) + expect(request).toHaveBeenCalledWith('inspectProcess', { + sessionId: 'sess-a', + steadyState: true + }) + adapter.dispose() + }) + + it('omits the field entirely when not requested, so the wire is byte-identical to before', async () => { + const request = vi.fn(async () => ({ foregroundProcess: null, hasChildProcesses: false })) + const adapter = createAdapter(PROTOCOL_VERSION, request) + await adapter.inspectProcess('sess-a', { expectedIncarnationId: 'inc-1', steadyState: false }) + expect(request).toHaveBeenCalledWith('inspectProcess', { + sessionId: 'sess-a', + expectedIncarnationId: 'inc-1' + }) + adapter.dispose() + }) + + it('an old daemon that ignores steadyState still answers with the full-capture shape, and the client accepts it', async () => { + // A pre-field daemon returns exactly what it always did: name + evidence, never a cheap answer. + const oldDaemonAnswer = { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + authorityGeneration: 'gen', + observationEpoch: 1, + capturedAgeMs: 0, + ptyId: 'sess-a', + ptyIncarnationId: 'inc-1' + } + } + const request = vi.fn(async () => oldDaemonAnswer) + const adapter = createAdapter(COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION, request) + await expect(adapter.inspectProcess('sess-a', { steadyState: true })).resolves.toEqual( + oldDaemonAnswer + ) + adapter.dispose() + }) + + it('a pre-inspection daemon never sees the field: the client composes from getForegroundProcess as before', async () => { + const request = vi.fn(async () => ({ foregroundProcess: 'codex' })) + const adapter = createAdapter(COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION - 1, request) + await expect(adapter.inspectProcess('sess-a', { steadyState: true })).resolves.toEqual({ + foregroundProcess: 'codex', + hasChildProcesses: true + }) + expect(request).toHaveBeenCalledWith('getForegroundProcess', { sessionId: 'sess-a' }) + adapter.dispose() + }) +}) diff --git a/src/main/daemon/daemon-pty-process-inspection.ts b/src/main/daemon/daemon-pty-process-inspection.ts index b05a8a6c6b6..335c641da62 100644 --- a/src/main/daemon/daemon-pty-process-inspection.ts +++ b/src/main/daemon/daemon-pty-process-inspection.ts @@ -25,7 +25,7 @@ export abstract class DaemonPtyProcessInspection extends DaemonPtyBufferSnapshot async inspectProcess( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { if (this.protocolVersion < GET_FOREGROUND_PROCESS_PROTOCOL_VERSION) { return clientOnlyUnverifiableInspection('old_host') @@ -47,7 +47,9 @@ export abstract class DaemonPtyProcessInspection extends DaemonPtyBufferSnapshot sessionId: id, ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } - : {}) + : {}), + // Additive: an older daemon ignores it and pays for the full capture. + ...(options?.steadyState === true ? { steadyState: true } : {}) }) } diff --git a/src/main/daemon/daemon-pty-router.ts b/src/main/daemon/daemon-pty-router.ts index 78e12215504..962cde6760e 100644 --- a/src/main/daemon/daemon-pty-router.ts +++ b/src/main/daemon/daemon-pty-router.ts @@ -179,7 +179,7 @@ export class DaemonPtyRouter implements IPtyProvider { async inspectProcess( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { return this.adapterForInspection(id).inspectProcess(id, options) } diff --git a/src/main/daemon/daemon-request-router.ts b/src/main/daemon/daemon-request-router.ts index e081281fa8a..bb7d0d1a256 100644 --- a/src/main/daemon/daemon-request-router.ts +++ b/src/main/daemon/daemon-request-router.ts @@ -105,12 +105,17 @@ export class DaemonRequestRouter { return { foregroundProcess: this.options.host.getForegroundProcess(request.payload.sessionId) } - case 'inspectProcess': - return request.payload.expectedIncarnationId - ? this.options.host.inspectProcess(request.payload.sessionId, { - expectedIncarnationId: request.payload.expectedIncarnationId - }) + case 'inspectProcess': { + const options = { + ...(request.payload.expectedIncarnationId + ? { expectedIncarnationId: request.payload.expectedIncarnationId } + : {}), + ...(request.payload.steadyState === true ? { steadyState: true } : {}) + } + return Object.keys(options).length > 0 + ? this.options.host.inspectProcess(request.payload.sessionId, options) : this.options.host.inspectProcess(request.payload.sessionId) + } case 'confirmForegroundProcess': return { foregroundProcess: await this.options.host.confirmForegroundProcess( diff --git a/src/main/daemon/pty-subprocess/foreground-process-tracker.ts b/src/main/daemon/pty-subprocess/foreground-process-tracker.ts index 115b795202d..8726dc87281 100644 --- a/src/main/daemon/pty-subprocess/foreground-process-tracker.ts +++ b/src/main/daemon/pty-subprocess/foreground-process-tracker.ts @@ -36,7 +36,9 @@ type CachedAgentForeground = { processName: string; pid: number | null; refreshe export type PtyForegroundProcessTracker = { recordOutput(data: string): void markDead(): void - getForegroundProcess(): string | null + /** `rawFallback`: node-pty's own name only, with no identity cache and no background + * process-table refresh -- the cheap-tier tick must not fork a full `ps` as a side effect. */ + getForegroundProcess(options?: { rawFallback?: boolean }): string | null confirmForegroundProcess(): Promise confirmShellForeground(): Promise } @@ -213,10 +215,13 @@ export function createPtyForegroundProcessTracker(args: { cachedAgentForeground = null startupAgentForeground = null }, - getForegroundProcess: () => { + getForegroundProcess: (options) => { if (args.isDead()) { return null } + if (options?.rawFallback === true) { + return getFallbackProcess() + } try { const fallbackProcess = getFallbackProcess() const fallbackRecognition = recognizeAgentProcess(fallbackProcess) diff --git a/src/main/daemon/session-subprocess-handle.ts b/src/main/daemon/session-subprocess-handle.ts index f14469afbb4..9268686d78e 100644 --- a/src/main/daemon/session-subprocess-handle.ts +++ b/src/main/daemon/session-subprocess-handle.ts @@ -6,7 +6,7 @@ export type SubprocessHandle = { pid: number /** Live foreground process name of the PTY (node-pty's `.process`), e.g. * 'claude' / 'codex' / 'zsh'. Null once the child has exited. */ - getForegroundProcess(): string | null + getForegroundProcess(options?: { rawFallback?: boolean }): string | null /** Await process-table evidence captured after this confirmation request. */ confirmForegroundProcess?(): Promise /** Proves a fresh post-boundary PTY process tree contains only the shell. */ diff --git a/src/main/daemon/session.ts b/src/main/daemon/session.ts index b0f1dfa538d..9265b2acfef 100644 --- a/src/main/daemon/session.ts +++ b/src/main/daemon/session.ts @@ -252,8 +252,8 @@ export class Session { return this.output.getCwd() } - getForegroundProcess(): string | null { - return this.subprocess.getForegroundProcess() + getForegroundProcess(options?: { rawFallback?: boolean }): string | null { + return this.subprocess.getForegroundProcess(options) } async confirmForegroundProcess(): Promise { diff --git a/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts b/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts new file mode 100644 index 00000000000..b02eec33773 --- /dev/null +++ b/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts @@ -0,0 +1,217 @@ +// Measurement for the cheap-tier process inspection. Drives the REAL daemon inspection +// entrypoint (`inspectTerminalHostProcess`) for 8 idle agent panes over a simulated 60s idle +// cadence (POLL_TIER_INTERVAL_MS.idle = 2,000ms) and counts `ps` forks BY COLUMN SET: a fork +// asking for `command=` is the full capture (measured 0.34-0.50s on a 1,900-process Mac, 1.15s +// on Linux), one without it is the cheap capture (0.03s on both). CI cannot time a real `ps` +// portably, so fork counts by column set are what this test measures; the per-fork costs above +// are the numbers measured by hand on the reference hosts. +// +// The second test is the zero-trade-off proof: the same tick sequence, including an agent exit +// and a restart, produces the identical foregroundProcess series with the cheap tier on and off. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Two seams because the two tiers spawn differently: the full evidence reader still forks +// through node:child_process, the cheap reader through Orca's runProcess entry point. +const { execFileMock, runProcessMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + runProcessMock: vi.fn() +})) +vi.mock('node:child_process', () => ({ execFile: execFileMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { resetCheapProcessTableSnapshotForTests } from '../../shared/cheap-process-table-snapshot-reader' +import { resetProcessTableSnapshotForTests } from '../../shared/process-table-snapshot-reader' +import { createPtyForegroundProcessTracker } from './pty-subprocess/foreground-process-tracker' +import { inspectTerminalHostProcess } from './terminal-host-process-inspection' +import type { Session } from './session' + +const PANE_COUNT = 8 +const IDLE_POLL_INTERVAL_MS = 2_000 // POLL_TIER_INTERVAL_MS.idle +const WINDOW_SECONDS = 60 +const TICKS = Math.floor((WINDOW_SECONDS * 1000) / IDLE_POLL_INTERVAL_MS) + +const shellPid = (pane: number): number => 1000 + pane * 100 +const agentPid = (pane: number): number => shellPid(pane) + 1 + +type PaneState = { agent: boolean; agentStart: string } +const panes: PaneState[] = Array.from({ length: PANE_COUNT }, () => ({ + agent: true, + agentStart: 'Thu Sep 3 16:02:05 2026' +})) + +const forks = { full: 0, cheap: 0 } + +function renderRows(): { full: string; cheap: string } { + const full: string[] = [] + const cheap: string[] = [] + panes.forEach((pane, i) => { + const s = shellPid(i) + const a = agentPid(i) + const tpgid = pane.agent ? a : s + const shellStat = pane.agent ? 'Ss' : 'Ss+' + cheap.push(`${s} 1 ${s} ${tpgid} ${shellStat} Thu Sep 3 16:02:01 2026`) + full.push(`${s} 1 ${s} ${tpgid} ${shellStat} ttys00${i} Thu Sep 3 16:02:01 2026 -zsh`) + if (pane.agent) { + cheap.push(`${a} ${s} ${a} ${a} S+ ${pane.agentStart}`) + full.push(`${a} ${s} ${a} ${a} S+ ttys00${i} ${pane.agentStart} node /usr/local/bin/claude`) + } + }) + return { full: `${full.join('\n')}\n`, cheap: `${cheap.join('\n')}\n` } +} + +function installCountingPs(): void { + execFileMock.mockImplementation((_cmd: string, args: string[], _opts: unknown, cb: unknown) => { + expect(args[1]).toContain('command=') + forks.full += 1 + ;(cb as (err: unknown, r: { stdout: string; stderr: string }) => void)(null, { + stdout: renderRows().full, + stderr: '' + }) + }) + runProcessMock.mockImplementation(async (spec: { args: readonly string[] }) => { + expect(spec.args[1]).not.toContain('command=') + forks.cheap += 1 + return { code: 0, signal: null, stdout: renderRows().cheap, stderr: '', timedOut: false } + }) +} + +function createSession(pane: number): Session { + const tracker = createPtyForegroundProcessTracker({ + process: { + pid: shellPid(pane), + get process() { + return panes[pane].agent ? 'node' : 'zsh' + } + } as never, + shellPath: '/bin/zsh', + sessionId: `wt:pane-${pane}`, + startupAgentRecognition: null, + isDead: () => false + }) + return { + pid: shellPid(pane), + incarnationId: `inc-${pane}`, + isAlive: true, + getForegroundProcess: (options?: { rawFallback?: boolean }) => + tracker.getForegroundProcess(options) + } as unknown as Session +} + +async function settle(): Promise { + for (let i = 0; i < 8; i += 1) { + await Promise.resolve() + } +} + +async function runTick(sessions: Session[], steadyState: boolean): Promise<(string | null)[]> { + const results = await Promise.all( + sessions.map((session, pane) => + inspectTerminalHostProcess({ + sessionId: `wt:pane-${pane}`, + session, + ...(steadyState ? { steadyState: true } : {}), + authorityGeneration: 'gen', + nextObservationEpoch: () => 1 + }) + ) + ) + await settle() + return results.map((r) => r.foregroundProcess) +} + +describe('cheap-tier ps scan volume at 8 idle agent panes over 60s', () => { + let platform: PropertyDescriptor | undefined + + beforeEach(() => { + execFileMock.mockReset() + runProcessMock.mockReset() + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + forks.full = 0 + forks.cheap = 0 + panes.forEach((pane) => { + pane.agent = true + pane.agentStart = 'Thu Sep 3 16:02:05 2026' + }) + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(1_000_000) + installCountingPs() + }) + + afterEach(() => { + vi.useRealTimers() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + it('replaces ~all full captures with cheap ones once every pane holds an anchor', async () => { + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + const names = await runTick(sessions, true) + expect(names.every((name) => name === 'claude')).toBe(true) + } + // Baseline today: one full capture per tick (TTL-shared across the 8 panes) = TICKS. + // Now: the first tick establishes every anchor from one full capture; every later tick is + // one TTL-shared cheap capture. Published numbers, from this run: + // before: 30 full (~0.34-0.50s each on macOS, 1.15s Linux) + 0 cheap + // after: 1 full + 29 cheap (~0.03s each) + expect(forks.full).toBe(1) + expect(forks.cheap).toBe(TICKS - 1) + expect(forks.full + forks.cheap).toBe(TICKS) + }) + + it('keeps today’s cost when the caller does not opt in (old client / remote / restore)', async () => { + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + await runTick(sessions, false) + } + expect(forks.cheap).toBe(0) + expect(forks.full).toBe(TICKS) + }) + + it('completion detection is byte-for-byte unchanged: exit, idle, and restart resolve identically with and without the cheap tier', async () => { + const script = async (steadyState: boolean): Promise<(string | null)[][]> => { + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + panes.forEach((pane) => { + pane.agent = true + pane.agentStart = 'Thu Sep 3 16:02:05 2026' + }) + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + const series: (string | null)[][] = [] + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + if (tick === 5) { + panes[2].agent = false // pane 2's agent exits + } + if (tick === 12) { + panes[2].agent = true // ...and is restarted with a new start time + panes[2].agentStart = 'Thu Sep 3 16:30:00 2026' + } + if (tick === 20) { + panes[6].agent = false + } + series.push(await runTick(sessions, steadyState)) + } + return series + } + const withCheapTier = await script(true) + const cheapForks = forks.cheap + forks.cheap = 0 + forks.full = 0 + const fullOnly = await script(false) + expect(withCheapTier).toEqual(fullOnly) + // And the exit was seen on the very tick it happened, in both modes. + expect(withCheapTier[4][2]).toBe('claude') + expect(withCheapTier[5][2]).not.toBe('claude') + expect(withCheapTier[12][2]).toBe('claude') + expect(withCheapTier[19][6]).toBe('claude') + expect(withCheapTier[20][6]).not.toBe('claude') + expect(cheapForks).toBeGreaterThan(0) + }) +}) diff --git a/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts b/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts new file mode 100644 index 00000000000..abf086e6e46 --- /dev/null +++ b/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts @@ -0,0 +1,297 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Two seams because the two tiers spawn differently: the full evidence reader still forks +// through node:child_process, the cheap reader through Orca's runProcess entry point. +const { execFileMock, runProcessMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + runProcessMock: vi.fn() +})) +vi.mock('node:child_process', () => ({ execFile: execFileMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { resetCheapProcessTableSnapshotForTests } from '../../shared/cheap-process-table-snapshot-reader' +import { resetProcessTableSnapshotForTests } from '../../shared/process-table-snapshot-reader' +import { createPtyForegroundProcessTracker } from './pty-subprocess/foreground-process-tracker' +import { + inspectTerminalHostProcess, + type TerminalHostInspectionTier +} from './terminal-host-process-inspection' +import { getSteadyStateAnchor } from './terminal-host-steady-state-anchor' +import type { Session } from './session' + +const SHELL_PID = 4242 +const AGENT_PID = 4300 +const START_SHELL = 'Thu Sep 3 16:02:01 2026' +const START_AGENT = 'Thu Sep 3 16:02:05 2026' + +type Table = { agent: 'claude' | 'stopped' | 'gone' | 'replaced'; children?: number } + +/** One host table rendered in both column sets, so each fork answers by the args it asked for. */ +function renderTable(table: Table): { full: string; cheap: string } { + const shellTpgid = table.agent === 'claude' || table.agent === 'replaced' ? AGENT_PID : SHELL_PID + const shellStat = shellTpgid === SHELL_PID ? 'Ss+' : 'Ss' + const rows: { cheap: string; full: string }[] = [ + { + cheap: `${SHELL_PID} 1 ${SHELL_PID} ${shellTpgid} ${shellStat} ${START_SHELL}`, + full: `${SHELL_PID} 1 ${SHELL_PID} ${shellTpgid} ${shellStat} ttys004 ${START_SHELL} -zsh` + }, + { + cheap: `9000 1 9000 9000 Ss+ Thu Sep 3 12:00:00 2026`, + full: `9000 1 9000 9000 Ss+ ttys009 Thu Sep 3 12:00:00 2026 -zsh` + } + ] + if (table.agent !== 'gone') { + const stat = table.agent === 'stopped' ? 'T' : 'S+' + const start = table.agent === 'replaced' ? 'Thu Sep 3 16:30:00 2026' : START_AGENT + rows.push({ + cheap: `${AGENT_PID} ${SHELL_PID} ${AGENT_PID} ${shellTpgid} ${stat} ${start}`, + full: `${AGENT_PID} ${SHELL_PID} ${AGENT_PID} ${shellTpgid} ${stat} ttys004 ${start} node /usr/local/bin/claude` + }) + for (let i = 0; i < (table.children ?? 0); i += 1) { + const pid = AGENT_PID + 10 + i + rows.push({ + cheap: `${pid} ${AGENT_PID} ${AGENT_PID} ${shellTpgid} S+ Thu Sep 3 16:05:0${i} 2026`, + full: `${pid} ${AGENT_PID} ${AGENT_PID} ${shellTpgid} S+ ttys004 Thu Sep 3 16:05:0${i} 2026 rg --files` + }) + } + } + return { + full: `${rows.map((r) => r.full).join('\n')}\n`, + cheap: `${rows.map((r) => r.cheap).join('\n')}\n` + } +} + +const forks = { full: 0, cheap: 0 } +let table: Table = { agent: 'claude' } + +function installPs(): void { + execFileMock.mockImplementation((_cmd: string, args: string[], _opts: unknown, cb: unknown) => { + expect(args[1]).toContain('command=') + forks.full += 1 + ;(cb as (err: unknown, r: { stdout: string; stderr: string }) => void)(null, { + stdout: renderTable(table).full, + stderr: '' + }) + }) + runProcessMock.mockImplementation(async (spec: { args: readonly string[] }) => { + expect(spec.args[1]).not.toContain('command=') + forks.cheap += 1 + return { + code: 0, + signal: null, + stdout: renderTable(table).cheap, + stderr: '', + timedOut: false + } + }) +} + +function createSession(processName: () => string): Session { + let dead = false + const tracker = createPtyForegroundProcessTracker({ + process: { + pid: SHELL_PID, + get process() { + return processName() + } + } as never, + shellPath: '/bin/zsh', + sessionId: 'wt-1:pane-1', + startupAgentRecognition: null, + isDead: () => dead + }) + return { + pid: SHELL_PID, + incarnationId: 'inc-1', + get isAlive() { + return !dead + }, + getForegroundProcess: (options?: { rawFallback?: boolean }) => + tracker.getForegroundProcess(options), + markDead: () => { + dead = true + tracker.markDead() + } + } as unknown as Session & { markDead(): void } +} + +async function inspect( + session: Session, + options: { steadyState?: boolean; expectedIncarnationId?: string } = {} +): Promise<{ + tier: TerminalHostInspectionTier + result: Awaited> +}> { + let tier: TerminalHostInspectionTier = 'full' + const result = await inspectTerminalHostProcess({ + sessionId: 'wt-1:pane-1', + session, + ...options, + authorityGeneration: 'gen-1', + nextObservationEpoch: () => 1, + onTier: (t) => { + tier = t + } + }) + return { tier, result } +} + +async function settle(): Promise { + // The tracker's recognizing refresh runs off the same TTL-shared capture; let it land. + for (let i = 0; i < 8; i += 1) { + await Promise.resolve() + } +} + +async function advance(ms: number): Promise { + vi.setSystemTime(Date.now() + ms) +} + +describe('daemon cheap-tier process inspection', () => { + let platform: PropertyDescriptor | undefined + + beforeEach(() => { + execFileMock.mockReset() + runProcessMock.mockReset() + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + forks.full = 0 + forks.cheap = 0 + table = { agent: 'claude' } + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(1_000_000) + installPs() + }) + + afterEach(() => { + vi.useRealTimers() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + /** Bring a session to a recognized anchor the way production does: one full cadence tick. */ + async function anchoredSession(): Promise { + const session = createSession(() => 'node') + const first = await inspect(session, { steadyState: true }) + await settle() + expect(first.tier).toBe('full') + expect(first.result.foregroundProcess).toBe('claude') + expect(getSteadyStateAnchor(session)?.agentName).toBe('claude') + return session + } + + it('a pane with NO recognized anchor never takes the cheap path, even when asked', async () => { + table = { agent: 'gone' } + const session = createSession(() => 'zsh') + for (let tick = 0; tick < 5; tick += 1) { + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toBeDefined() + } + expect(forks.cheap).toBe(0) + expect(forks.full).toBe(5) + }) + + it('serves an unchanged anchored pane from the cheap tier and OMITS evidence rather than faking it', async () => { + const session = await anchoredSession() + const fullBefore = forks.full + for (let tick = 0; tick < 4; tick += 1) { + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('cheap') + expect(result.foregroundProcess).toBe('claude') + expect(result.hasChildProcesses).toBe(true) + expect(result).not.toHaveProperty('foregroundProcessEvidence') + } + expect(forks.cheap).toBe(4) + expect(forks.full).toBe(fullBefore) + }) + + it('a request without steadyState (old client, remote client, restore path) always gets the full capture with evidence', async () => { + const session = await anchoredSession() + await advance(2_000) + const { tier, result } = await inspect(session) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toMatchObject({ + verdict: 'live', + processName: 'claude' + }) + expect(forks.cheap).toBe(0) + }) + + it('escalates to the full capture the moment the agent exits, and reports the exit', async () => { + const session = await anchoredSession() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + table = { agent: 'gone' } + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toMatchObject({ verdict: 'live', processName: null }) + }) + + it.each<[string, Table]>([ + ['Ctrl-Z stops the agent', { agent: 'stopped' }], + ['exit-and-replace reuses the pid', { agent: 'replaced' }], + ['a child spawns under the agent', { agent: 'claude', children: 1 }] + ])('escalates when %s', async (_name, next) => { + const session = await anchoredSession() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + table = next + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + }) + + it('escalates when node-pty reports a different foreground name, without waiting on ps', async () => { + let name = 'node' + const session = createSession(() => name) + await inspect(session, { steadyState: true }) + await settle() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + name = 'zsh' + await advance(2_000) + const cheapBefore = forks.cheap + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + expect(forks.cheap).toBe(cheapBefore) + }) + + it('falls through to the full capture when the cheap fork fails, and after an incarnation mismatch', async () => { + const session = await anchoredSession() + await advance(2_000) + runProcessMock.mockRejectedValueOnce(new Error('ps died')) + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + await advance(2_000) + const mismatched = await inspect(session, { steadyState: true, expectedIncarnationId: 'other' }) + expect(mismatched.tier).toBe('full') + expect(mismatched.result.foregroundProcessEvidence).toMatchObject({ + reason: 'incarnation_mismatch' + }) + }) + + it('a dead session is never served from its anchor', async () => { + const session = (await anchoredSession()) as Session & { markDead(): void } + session.markDead() + await expect(inspect(session, { steadyState: true })).rejects.toThrow('not found') + expect(forks.cheap).toBe(0) + }) + + it('an anchor is dropped when a full capture stops naming a recognized agent', async () => { + const session = await anchoredSession() + table = { agent: 'gone' } + await advance(2_000) + await inspect(session, { steadyState: true }) + expect(getSteadyStateAnchor(session)).toBeNull() + // Back with a new agent, but the pane must re-anchor via a FULL capture first. + table = { agent: 'claude' } + await advance(2_000) + const cheapBefore = forks.cheap + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + expect(forks.cheap).toBe(cheapBefore) + }) +}) diff --git a/src/main/daemon/terminal-host-process-inspection.ts b/src/main/daemon/terminal-host-process-inspection.ts index 181ca7266d7..6810687631e 100644 --- a/src/main/daemon/terminal-host-process-inspection.ts +++ b/src/main/daemon/terminal-host-process-inspection.ts @@ -1,8 +1,15 @@ import { isShellProcess } from '../../shared/agent-detection' import type { RemoteForegroundEvidence } from '../../shared/foreground-process-evidence' +import { getCheapProcessTableSnapshot } from '../../shared/cheap-process-table-snapshot-reader' import { getStrictProcessTableSnapshotWithAge } from '../../shared/process-table-snapshot-reader' import { resolveRemoteForegroundEvidence } from '../providers/agent-foreground-process' +import { buildPaneProcessFingerprint } from '../providers/posix-pane-foreground-fingerprint' import type { Session } from './session' +import { + clearSteadyStateAnchor, + getSteadyStateAnchor, + rememberSteadyStateAnchor +} from './terminal-host-steady-state-anchor' import { SessionNotFoundError } from './types' export type TerminalHostProcessInspection = { @@ -13,13 +20,23 @@ export type TerminalHostProcessInspection = { type RetiredIncarnation = { incarnationId: string; code: number; expiresAt: number } +/** + * Tick tiers for a POSIX pane. `cheap` forks `ps` without `tty=`/`command=` (11-38x cheaper) + * and answers from the anchored identity when the pane fingerprint is unchanged; anything it + * cannot prove escalates to `full`, today's evidence capture. + */ +export type TerminalHostInspectionTier = 'full' | 'cheap' + export async function inspectTerminalHostProcess(args: { sessionId: string session: Session | null expectedIncarnationId?: string + /** The caller is a self-correcting poll that only reads the process name, never evidence. */ + steadyState?: boolean retiredIncarnation?: RetiredIncarnation authorityGeneration: string nextObservationEpoch: () => number + onTier?: (tier: TerminalHostInspectionTier) => void }): Promise { const { sessionId, session, expectedIncarnationId, retiredIncarnation } = args if (!session || !session.isAlive) { @@ -45,9 +62,22 @@ export async function inspectTerminalHostProcess(args: { throw new SessionNotFoundError(sessionId) } + const incarnationMatches = + !expectedIncarnationId || expectedIncarnationId === session.incarnationId + if (args.steadyState === true && incarnationMatches) { + const anchored = await readAnchoredForeground(session) + if (anchored !== null) { + args.onTier?.('cheap') + // No evidence member on purpose: a tty-less capture cannot fence anything, and a + // fabricated fence would be read by remote/restore consumers as an observation. + return { foregroundProcess: anchored, hasChildProcesses: true } + } + } + args.onTier?.('full') + const foregroundProcess = session.getForegroundProcess() let evidence: RemoteForegroundEvidence - if (expectedIncarnationId && expectedIncarnationId !== session.incarnationId) { + if (!incarnationMatches) { evidence = unverifiableEvidence(args, session, 'incarnation_mismatch') } else { try { @@ -64,8 +94,10 @@ export async function inspectTerminalHostProcess(args: { }, snapshot.rows ) + await rememberSteadyStateAnchor(session, evidence, snapshot.rows) } catch { evidence = unverifiableEvidence(args, session, 'process_table_unreadable') + clearSteadyStateAnchor(session) } } return { @@ -75,6 +107,32 @@ export async function inspectTerminalHostProcess(args: { } } +/** + * Cheap tier, gated on an anchor the last full capture established. Start discovery therefore + * keeps today's exact behaviour: a pane with no anchor never gets here. A recognized agent's + * exit is a pid vanishing from the subtree, which the fingerprint always sees, so completion + * detection is unaffected. Any mismatch, unreadable capture, changed node-pty name, or non-POSIX + * host answers null -> full tier. + */ +async function readAnchoredForeground(session: Session): Promise { + const anchor = getSteadyStateAnchor(session) + if (process.platform === 'win32' || !anchor) { + return null + } + if (session.getForegroundProcess({ rawFallback: true }) !== anchor.rawFallback) { + return null + } + try { + const observed = await buildPaneProcessFingerprint( + await getCheapProcessTableSnapshot(), + session.pid + ) + return observed !== null && observed === anchor.fingerprint ? anchor.agentName : null + } catch { + return null + } +} + function unverifiableEvidence( args: { sessionId: string diff --git a/src/main/daemon/terminal-host-steady-state-anchor.ts b/src/main/daemon/terminal-host-steady-state-anchor.ts new file mode 100644 index 00000000000..562224f856f --- /dev/null +++ b/src/main/daemon/terminal-host-steady-state-anchor.ts @@ -0,0 +1,58 @@ +import { recognizeAgentProcess } from '../../shared/agent-process-recognition' +import type { RemoteForegroundEvidence } from '../../shared/foreground-process-evidence' +import { buildPaneProcessFingerprint } from '../providers/posix-pane-foreground-fingerprint' +import type { Session } from './session' + +/** + * What the last FULL capture proved about a pane: a recognized agent name, the pane subtree + * fingerprint at that moment, and node-pty's raw foreground name at that moment. A later cheap + * tick may re-serve `agentName` only while both of the latter still match. + */ +export type SteadyStateAnchor = { + agentName: string + fingerprint: string + rawFallback: string | null +} + +// Weakly keyed: an anchor dies with its Session, and a recycled pid under a new Session can +// never inherit one. Retired sessions fail `isAlive` before any read gets here regardless. +const anchors = new WeakMap() + +export function getSteadyStateAnchor(session: Session): SteadyStateAnchor | null { + return anchors.get(session) ?? null +} + +export function clearSteadyStateAnchor(session: Session): void { + anchors.delete(session) +} + +/** + * Record (or drop) the anchor after a full capture. Only a `live` verdict naming a recognized + * agent establishes one: the cheap tier is licensed by proven identity, never by a fallback name + * or an unverifiable read, so a pane without one always pays for the full capture. + */ +export async function rememberSteadyStateAnchor( + session: Session, + evidence: RemoteForegroundEvidence, + rows: Parameters[0] +): Promise { + if (evidence.verdict !== 'live' || !recognizeAgentProcess(evidence.processName)) { + anchors.delete(session) + return + } + let fingerprint: string | null + try { + fingerprint = await buildPaneProcessFingerprint(rows, session.pid) + } catch { + fingerprint = null + } + if (fingerprint === null || evidence.processName === null) { + anchors.delete(session) + return + } + anchors.set(session, { + agentName: evidence.processName, + fingerprint, + rawFallback: session.getForegroundProcess({ rawFallback: true }) + }) +} diff --git a/src/main/daemon/terminal-host.ts b/src/main/daemon/terminal-host.ts index 18f7b82a90f..95bedd1a7fd 100644 --- a/src/main/daemon/terminal-host.ts +++ b/src/main/daemon/terminal-host.ts @@ -233,7 +233,7 @@ export class TerminalHost { inspectProcess( sessionId: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { pruneRetiredPtyIncarnations(this.retiredIncarnations) const session = this.sessions.get(sessionId) @@ -253,6 +253,7 @@ export class TerminalHost { ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } : {}), + ...(options?.steadyState === true ? { steadyState: true } : {}), retiredIncarnation: this.retiredIncarnations.get(sessionId), authorityGeneration: this.authorityGeneration, nextObservationEpoch: () => ++this.observationEpoch diff --git a/src/main/ipc/pty/ipc/inspect.ts b/src/main/ipc/pty/ipc/inspect.ts index 041abaa7854..5a6d03d8855 100644 --- a/src/main/ipc/pty/ipc/inspect.ts +++ b/src/main/ipc/pty/ipc/inspect.ts @@ -172,7 +172,12 @@ export function installPtyInspectIpcHandlers(deps: { 'pty:inspectProcess', async ( _event, - args: { id: string; expectedIncarnationId?: string; scanChildProcesses?: boolean } + args: { + id: string + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ) => { // Why: same routing hazard as pty:hasPty — an unroutable id must read as client-only unverifiable, not as a local-provider answer or a raised IPC error. if (typeof args?.id !== 'string' || !args.id || args.id.startsWith('remote:')) { @@ -189,7 +194,8 @@ export function installPtyInspectIpcHandlers(deps: { ...(args.expectedIncarnationId ? { expectedIncarnationId: args.expectedIncarnationId } : {}), - ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}), + ...(args.steadyState === true ? { steadyState: true } : {}) } return Object.keys(options).length > 0 ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, options) diff --git a/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts b/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts new file mode 100644 index 00000000000..37d4cf3c4d0 --- /dev/null +++ b/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts @@ -0,0 +1,138 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as ProcessTableSnapshotReader from '../../shared/process-table-snapshot-reader' + +const { cheapSnapshotMock, fullSnapshotMock, resolveMock } = vi.hoisted(() => ({ + cheapSnapshotMock: vi.fn(), + fullSnapshotMock: vi.fn(), + resolveMock: vi.fn() +})) + +vi.mock('../../shared/cheap-process-table-snapshot-reader', () => ({ + getCheapProcessTableSnapshot: cheapSnapshotMock +})) +vi.mock('../../shared/process-table-snapshot-reader', async (importOriginal) => ({ + ...(await importOriginal()), + getProcessTableSnapshot: fullSnapshotMock +})) +vi.mock('./agent-foreground-process', () => ({ + resolveAgentForegroundProcessWithAvailability: resolveMock, + confirmShellForegroundProcess: vi.fn() +})) + +import { getLocalPtyForegroundProcess } from './local-pty-foreground-inspection' +import { ptyLastRecognizedForeground, ptyProcesses, ptyShellName } from './local-pty-provider-state' + +const SHELL_PID = 4242 +const AGENT_PID = 4300 +const ID = 'pty-1' + +type Table = 'agent' | 'shell-only' +let table: Table = 'agent' + +function rows(): Record[] { + const tpgid = table === 'agent' ? AGENT_PID : SHELL_PID + const out: Record[] = [ + { + pid: SHELL_PID, + ppid: 1, + pgid: SHELL_PID, + tpgid, + stat: table === 'agent' ? 'Ss' : 'Ss+', + tty: 'ttys004', + startTime: 'Thu Sep 3 16:02:01 2026', + command: '-zsh' + } + ] + if (table === 'agent') { + out.push({ + pid: AGENT_PID, + ppid: SHELL_PID, + pgid: AGENT_PID, + tpgid, + stat: 'S+', + tty: 'ttys004', + startTime: 'Thu Sep 3 16:02:05 2026', + command: 'node /usr/local/bin/claude' + }) + } + return out +} + +describe('local POSIX provider cheap-tier revalidation', () => { + let platform: PropertyDescriptor | undefined + const proc = { pid: SHELL_PID, process: 'node' } + + beforeEach(() => { + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + table = 'agent' + proc.process = 'node' + cheapSnapshotMock.mockReset() + cheapSnapshotMock.mockImplementation(async () => rows()) + fullSnapshotMock.mockReset() + fullSnapshotMock.mockImplementation(async () => rows()) + resolveMock.mockReset() + resolveMock.mockImplementation(async () => ({ + available: true, + processName: table === 'agent' ? 'claude' : 'zsh' + })) + ptyProcesses.set(ID, proc as never) + ptyShellName.set(ID, 'zsh') + ptyLastRecognizedForeground.delete(ID) + }) + + afterEach(() => { + ptyProcesses.delete(ID) + ptyShellName.delete(ID) + ptyLastRecognizedForeground.delete(ID) + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + it('a pane with NO recognized anchor never consults the cheap tier', async () => { + table = 'shell-only' + proc.process = 'zsh' + for (let i = 0; i < 3; i += 1) { + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + } + expect(cheapSnapshotMock).not.toHaveBeenCalled() + expect(resolveMock).toHaveBeenCalledTimes(3) + expect(ptyLastRecognizedForeground.get(ID)).toBeUndefined() + }) + + it('once recognized, an unchanged pane re-proves the agent from the cheap tier without a full scan', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + expect(resolveMock).toHaveBeenCalledTimes(1) + expect(ptyLastRecognizedForeground.get(ID)?.steady?.fingerprint).toEqual(expect.any(String)) + for (let i = 0; i < 3; i += 1) { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + } + expect(cheapSnapshotMock).toHaveBeenCalledTimes(3) + expect(resolveMock).toHaveBeenCalledTimes(1) + }) + + it('an agent exit changes the fingerprint, escalates to the full scan, and clears the anchor', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + table = 'shell-only' + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + expect(resolveMock).toHaveBeenCalledTimes(2) + expect(ptyLastRecognizedForeground.get(ID)).toBeUndefined() + }) + + it('a changed node-pty foreground name escalates without consulting the cheap tier', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + proc.process = 'zsh' + table = 'shell-only' + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + expect(cheapSnapshotMock).not.toHaveBeenCalled() + expect(resolveMock).toHaveBeenCalledTimes(2) + }) + + it('a cheap capture failure falls through to the full scan', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + cheapSnapshotMock.mockRejectedValueOnce(new Error('ps died')) + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + expect(resolveMock).toHaveBeenCalledTimes(2) + }) +}) diff --git a/src/main/providers/local-pty-foreground-inspection.ts b/src/main/providers/local-pty-foreground-inspection.ts index c42c06d9121..d4a717a9de2 100644 --- a/src/main/providers/local-pty-foreground-inspection.ts +++ b/src/main/providers/local-pty-foreground-inspection.ts @@ -1,8 +1,11 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' +import { getCheapProcessTableSnapshot } from '../../shared/cheap-process-table-snapshot-reader' +import { getProcessTableSnapshot } from '../../shared/process-table-snapshot-reader' import { confirmShellForegroundProcess, resolveAgentForegroundProcessWithAvailability } from './agent-foreground-process' +import { buildPaneProcessFingerprint } from './posix-pane-foreground-fingerprint' import { resolveForegroundFallbackProcess } from './local-pty-launch-helpers' import { ptyAgentForegroundContextPaths, @@ -35,6 +38,34 @@ export async function hasLocalPtyChildProcesses(id: string): Promise { } } +/** + * POSIX twin of the Windows job-membership short-circuit below: a pane that already holds a + * recognized agent re-proves it from the cheap `ps` tier when the subtree fingerprint is + * unchanged. Panes with no anchor never get here, so start discovery is untouched. + */ +async function revalidateCachedPosixAgent( + proc: { pid: number }, + cachedEntry: { + name: string + steady?: { fingerprint: string; fallbackProcess: string | null } | null + }, + fallbackProcess: string | null +): Promise { + const steady = cachedEntry.steady + if (!steady || steady.fallbackProcess !== fallbackProcess) { + return false + } + try { + const observed = await buildPaneProcessFingerprint( + await getCheapProcessTableSnapshot(), + proc.pid + ) + return observed !== null && observed === steady.fingerprint + } catch { + return false + } +} + export async function getLocalPtyForegroundProcess(id: string): Promise { const proc = ptyProcesses.get(id) if (!proc) { @@ -89,6 +120,18 @@ export async function getLocalPtyForegroundProcess(id: string): Promise { + if (process.platform === 'win32') { + return null + } + try { + const fingerprint = await buildPaneProcessFingerprint(await getProcessTableSnapshot(), shellPid) + return fingerprint === null ? null : { fingerprint, fallbackProcess } + } catch { + return null + } +} + export async function confirmLocalPtyForegroundProcess(id: string): Promise { const proc = ptyProcesses.get(id) if (!proc) { diff --git a/src/main/providers/local-pty-provider-state.ts b/src/main/providers/local-pty-provider-state.ts index 5e87502b1f6..9e54ac859b0 100644 --- a/src/main/providers/local-pty-provider-state.ts +++ b/src/main/providers/local-pty-provider-state.ts @@ -43,9 +43,16 @@ export const ptyAgentForegroundContextPaths = new Map() // Why: remember the last recognized agent foreground so a degraded scan doesn't report the shell and look like an exit. // `pid` anchors the identity to the row that proved it (null when ambiguous); // `at` is the last confirmation, so unanchored job evidence -- only a superset -- cannot hold it forever. +// `steady` (POSIX) is the pane fingerprint the recognizing capture proved plus node-pty's name at +// that moment; a cheap capture matching it re-proves the identity without the full table. export const ptyLastRecognizedForeground = new Map< string, - { name: string; pid: number | null; at: number } + { + name: string + pid: number | null + at: number + steady?: { fingerprint: string; fallbackProcess: string | null } | null + } >() export const ptyTerminalHandle = new Map() export const ptyWorktreeId = new Map() diff --git a/src/main/providers/posix-pane-foreground-fingerprint.test.ts b/src/main/providers/posix-pane-foreground-fingerprint.test.ts new file mode 100644 index 00000000000..76cb68f79dc --- /dev/null +++ b/src/main/providers/posix-pane-foreground-fingerprint.test.ts @@ -0,0 +1,187 @@ +import { describe, expect, it } from 'vitest' +import { + buildPaneProcessFingerprint, + type PaneFingerprintRow +} from './posix-pane-foreground-fingerprint' + +const SHELL = 4242 +const AGENT = 4300 +const OTHER_PANE = 9000 + +type Row = PaneFingerprintRow + +const shell = (over: Partial = {}): Row => ({ + pid: SHELL, + ppid: 1, + pgid: SHELL, + tpgid: AGENT, + stat: 'Ss', + startTime: 'Thu Sep 3 16:02:01 2026', + ...over +}) +const agent = (over: Partial = {}): Row => ({ + pid: AGENT, + ppid: SHELL, + pgid: AGENT, + tpgid: AGENT, + stat: 'S+', + startTime: 'Thu Sep 3 16:02:05 2026', + ...over +}) +const child = (pid: number, ppid: number, over: Partial = {}): Row => ({ + pid, + ppid, + pgid: AGENT, + tpgid: AGENT, + stat: 'S+', + startTime: `Thu Sep 3 16:03:${String(pid % 60).padStart(2, '0')} 2026`, + ...over +}) +const foreign = (): Row => ({ + pid: OTHER_PANE, + ppid: 1, + pgid: OTHER_PANE, + tpgid: OTHER_PANE, + stat: 'Ss+', + startTime: 'Thu Sep 3 12:00:00 2026' +}) + +const fp = (rows: Row[]): Promise => + buildPaneProcessFingerprint(rows, SHELL, { platform: 'darwin' }) + +describe('buildPaneProcessFingerprint', () => { + const baseline = [foreign(), shell(), agent()] + + it('is stable across captures that differ only in scheduler state, row order, and foreign panes', async () => { + const a = await fp(baseline) + expect(a).not.toBeNull() + // R vs S: a working agent flips this every tick and it says nothing about the pane. + expect(await fp([agent({ stat: 'R+' }), shell({ stat: 'Ss' }), foreign()])).toBe(a) + // The shell going idle-vs-runnable, or a foreign pane starting/exiting, is not our business. + expect(await fp([shell({ stat: 'Rs' }), agent()])).toBe(a) + // lstart padding differs between column sets; both must stamp identically. + expect(await fp([shell({ startTime: 'Thu Sep 3 16:02:01 2026' }), agent()])).toBe(a) + }) + + describe('escalates (fingerprint changes) on every completion-relevant transition', () => { + it('agent exit: the recognized pid vanishes from the subtree', async () => { + const before = await fp(baseline) + expect(await fp([foreign(), shell({ tpgid: SHELL, stat: 'Ss+' })])).not.toBe(before) + }) + + it('exit-and-replace: the same pid is reused by a new process with a new start time', async () => { + const before = await fp(baseline) + expect(await fp([shell(), agent({ startTime: 'Thu Sep 3 16:09:00 2026' })])).not.toBe(before) + }) + + it('Ctrl-Z: the agent stops and the shell takes the terminal back', async () => { + const before = await fp(baseline) + expect(await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'T' })])).not.toBe( + before + ) + }) + + it('bg: the stopped job resumes in the background, foreground stays with the shell', async () => { + const stopped = await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'T' })]) + const backgrounded = await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'S' })]) + expect(backgrounded).not.toBe(stopped) + expect(backgrounded).not.toBe(await fp(baseline)) + }) + + it('child churn: a subprocess appearing or disappearing under the agent', async () => { + const before = await fp(baseline) + const withChild = await fp([shell(), agent(), child(4310, AGENT)]) + expect(withChild).not.toBe(before) + expect(await fp([shell(), agent(), child(4310, AGENT), child(4311, 4310)])).not.toBe( + withChild + ) + // A child exec'ing away from the group (setsid / disown) is also a change. + expect(await fp([shell(), agent(), child(4310, AGENT, { pgid: 4310 })])).not.toBe(withChild) + }) + + it('shell replaced: same pid, different start time', async () => { + const before = await fp(baseline) + expect(await fp([shell({ startTime: 'Thu Sep 3 17:00:00 2026' }), agent()])).not.toBe(before) + }) + }) + + describe('refuses to fingerprint an unfenced pane (caller must take the full capture)', () => { + it('root shell missing from the capture', async () => { + expect(await fp([foreign(), agent()])).toBeNull() + }) + + it('root shell has no start marker', async () => { + expect(await fp([shell({ startTime: undefined }), agent()])).toBeNull() + }) + + it('root shell has no job-control columns', async () => { + expect(await fp([shell({ pgid: undefined, tpgid: undefined }), agent()])).toBeNull() + }) + }) + + describe('Linux', () => { + it('reads /proc start times for the pane subtree only and ignores ps start markers', async () => { + const asked: number[] = [] + const read = async (pid: number): Promise => { + asked.push(pid) + return pid === SHELL ? '1000' : pid === AGENT ? '2000' : null + } + const rows = [foreign(), shell({ startTime: undefined }), agent({ startTime: undefined })] + const a = await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: read + }) + expect(a).toContain(`${SHELL}@1000`) + expect(a).toContain(`${AGENT}@2000`) + expect(asked.sort()).toEqual([SHELL, AGENT].sort()) + // An exit-and-replace changes only the /proc start ticks. + const replaced = await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: async (pid) => (pid === AGENT ? '2500' : read(pid)) + }) + expect(replaced).not.toBe(a) + }) + + it('refuses when the root /proc entry cannot be read', async () => { + expect( + await buildPaneProcessFingerprint([shell(), agent()], SHELL, { + platform: 'linux', + readLinuxStartTime: async () => null + }) + ).toBeNull() + }) + it('refuses when a DESCENDANT start marker cannot be read', async () => { + // Why: the start marker is what makes a pid comparison recycle-safe. Stamping a missing + // one as a placeholder let two captures that both failed to read it compare equal across + // a recycled pid, so a vanished agent looked unchanged and the cheap tier kept serving + // its name. Refusing sends the caller to the full capture. + const rows = [shell(), agent()] + + expect( + await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: async (pid) => (pid === SHELL ? '2400' : null) + }) + ).toBeNull() + }) + + it('does not let a recycled descendant pid reuse a fingerprint', async () => { + // Both captures fail to read the descendant marker; the pid is reused by a different + // process in between. Equal fingerprints here would mask the agent's exit. + const readNoDescendant = async (pid: number): Promise => + pid === SHELL ? '2400' : null + const before = await buildPaneProcessFingerprint([shell(), agent()], SHELL, { + platform: 'linux', + readLinuxStartTime: readNoDescendant + }) + const after = await buildPaneProcessFingerprint( + [shell(), agent({ stat: 'S+', pgid: AGENT })], + SHELL, + { platform: 'linux', readLinuxStartTime: readNoDescendant } + ) + + expect(before).toBeNull() + expect(after).toBeNull() + }) + }) +}) diff --git a/src/main/providers/posix-pane-foreground-fingerprint.ts b/src/main/providers/posix-pane-foreground-fingerprint.ts new file mode 100644 index 00000000000..ac1539300e2 --- /dev/null +++ b/src/main/providers/posix-pane-foreground-fingerprint.ts @@ -0,0 +1,93 @@ +import { readFile } from 'node:fs/promises' +import { collectDescendantsFromIndex, getProcessTableIndex } from '../../shared/process-table-index' +import { parseLinuxProcStatStartTime } from '../../shared/process-table-snapshot-reader' + +/** The job-control columns both `ps` tiers carry; `command`/`tty` are deliberately absent. */ +export type PaneFingerprintRow = { + pid: number + ppid: number + pgid?: number + tpgid?: number + stat: string + startTime?: string +} + +export type PaneFingerprintDeps = { + platform?: NodeJS.Platform + /** Linux: `/proc//stat` field 22, read for the pane subtree only. */ + readLinuxStartTime?: (pid: number) => Promise +} + +/** + * Only the job-control bits of `stat`. The scheduler letter (R/S/D/I/U) flips every tick + * on a working agent and says nothing about whether the pane changed hands; stopped, + * zombie, and foreground-group membership do. + */ +function jobControlState(stat: string): string { + const head = stat[0] ?? '' + const lifecycle = head === 'T' || head === 't' ? 'T' : head === 'Z' ? 'Z' : '' + return lifecycle + (stat.includes('+') ? '+' : '') +} + +async function readLinuxProcStartTime(pid: number): Promise { + try { + return parseLinuxProcStatStartTime(await readFile(`/proc/${pid}/stat`, 'utf8')) + } catch { + return null + } +} + +/** + * A per-pane summary of everything the cheap `ps` tier can see: the root shell's identity + * (pid + start marker) and terminal foreground group, and every descendant's identity, group, + * and job-control state. Two captures with equal fingerprints describe the same pane + * subtree, so the name resolved from the last full capture still holds. + * + * Null when the root is missing or unfenced (no start marker, no group columns): callers + * must then take the full capture rather than trust a comparison that could not be made. + */ +export async function buildPaneProcessFingerprint( + rows: readonly PaneFingerprintRow[], + rootPid: number, + deps: PaneFingerprintDeps = {} +): Promise { + const platform = deps.platform ?? process.platform + const index = getProcessTableIndex(rows) + const root = index.byPid.get(rootPid) + if (!root || root.pgid === undefined || root.tpgid === undefined) { + return null + } + const descendants = collectDescendantsFromIndex(index, rootPid) + const subtree = [root, ...descendants] + let startTimes: ReadonlyMap + if (platform === 'linux') { + const read = deps.readLinuxStartTime ?? readLinuxProcStartTime + const entries = await Promise.all( + subtree.map(async (row) => [row.pid, await read(row.pid)] as const) + ) + startTimes = new Map(entries) + } else { + // Collapse `lstart` padding (`Sep 3`) so both column sets stamp identically. + startTimes = new Map( + subtree.map((row) => [row.pid, row.startTime?.replace(/\s+/g, ' ') ?? null] as const) + ) + } + const rootStart = startTimes.get(rootPid) + if (!rootStart) { + return null + } + // Why every member, not just the root: a start marker is what makes a pid comparison + // recycle-safe. Stamping a missing one as a placeholder would let two captures that both + // failed to read it compare equal across a recycled pid, so a vanished agent could look + // unchanged. Refusing the fingerprint sends the caller to the full capture instead. + const members: string[] = [] + for (const row of descendants) { + const startTime = startTimes.get(row.pid) + if (!startTime) { + return null + } + members.push(`${row.pid}@${startTime}:${row.pgid ?? '?'}:${jobControlState(row.stat)}`) + } + members.sort() + return `${rootPid}@${rootStart}#${root.tpgid}:${jobControlState(root.stat)}|${members.join(',')}` +} diff --git a/src/main/providers/pty-process-inspection.ts b/src/main/providers/pty-process-inspection.ts index 59d910b2238..2c316ae05f9 100644 --- a/src/main/providers/pty-process-inspection.ts +++ b/src/main/providers/pty-process-inspection.ts @@ -23,6 +23,9 @@ type CompletionSensitivePtyProvider = IPtyProvider & { export type PtyProcessInspectionOptions = { expectedIncarnationId?: PtyIncarnationId scanChildProcesses?: boolean + /** A self-correcting cadence poll that reads only the process name: licenses a host to answer + * from a cheap capture and OMIT evidence. Never set by a caller that consumes evidence. */ + steadyState?: boolean } export async function inspectPtyProviderProcess( diff --git a/src/preload/api/pty-api.ts b/src/preload/api/pty-api.ts index a1850398357..dff2b0b5185 100644 --- a/src/preload/api/pty-api.ts +++ b/src/preload/api/pty-api.ts @@ -112,7 +112,11 @@ export type PtyApi = { getForegroundProcess: (id: string) => Promise inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ) => Promise confirmForegroundProcess: (id: string) => Promise getCwd: (id: string) => Promise diff --git a/src/preload/api/pty-bridge-stream-and-serialization.ts b/src/preload/api/pty-bridge-stream-and-serialization.ts index 414a5514bfa..0847291ba7e 100644 --- a/src/preload/api/pty-bridge-stream-and-serialization.ts +++ b/src/preload/api/pty-bridge-stream-and-serialization.ts @@ -7,7 +7,11 @@ import type { TerminalProcessInspection } from '../../shared/terminal-process-in export const ptyStreamAndSerializationApi = { inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ): Promise => ipcRenderer.invoke('pty:inspectProcess', { id, ...options }), confirmForegroundProcess: (id: string): Promise => diff --git a/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts b/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts index 7d7c53f6bb8..7bf7e90f7f9 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts @@ -32,7 +32,7 @@ export type AgentCompletionCoordinatorOptions = { inspectProcess: ( settings: Pick | null | undefined, ptyId: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ) => Promise dispatchCompletion: (title: string, meta?: AgentCompletionDispatchMeta) => void dispatchAttention?: (title: string, meta: AgentAttentionDispatchMeta) => void diff --git a/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts b/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts index f34ceffa97e..ca2456aedbb 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts @@ -108,10 +108,18 @@ export function createAgentCompletionProcessMonitor({ let inspectedRecognizedAgent = false let inspectionSucceeded = false try { - const result = await (expectedIncarnationIdAtRequest - ? options.inspectProcess(options.getSettings(), ptyId, { - expectedIncarnationId: expectedIncarnationIdAtRequest - }) + // Only a cadence tick on a local pane reads nothing but the name; every other read + // (pending-title, remote) needs the full capture and must not ask for the cheap one. + const inspectOptions = { + ...(expectedIncarnationIdAtRequest + ? { expectedIncarnationId: expectedIncarnationIdAtRequest } + : {}), + ...(priority === 'cadence' && options.isRemotePtyId?.(ptyId) !== true + ? { steadyState: true } + : {}) + } + const result = await (Object.keys(inspectOptions).length > 0 + ? options.inspectProcess(options.getSettings(), ptyId, inspectOptions) : options.inspectProcess(options.getSettings(), ptyId)) if ( !state.disposed && diff --git a/src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts b/src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts new file mode 100644 index 00000000000..4f2286c7eb7 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts @@ -0,0 +1,84 @@ +import { describe, expect, it, vi } from 'vitest' +import { createAgentCompletionCoordinator } from './agent-completion-coordinator' +import { + flushAsyncTicks, + processResult, + useAgentCompletionCoordinatorLifecycle +} from './agent-completion-coordinator-test-harness' + +// The renderer opts a read into the cheap tier ONLY when it is a self-correcting cadence poll on a +// local pane. Pending-title reads decide a completion once and remote reads consume evidence, so +// neither may ask for a capture that omits evidence. +describe('agent completion steadyState opt-in', () => { + useAgentCompletionCoordinatorLifecycle() + + const optionsOf = (call: unknown[]): unknown => call[2] + + it('marks cadence polls on a local pane as steadyState', async () => { + const inspectProcess = vi.fn(async () => processResult('codex')) + const coordinator = createAgentCompletionCoordinator({ + paneKey: 'tab-1:leaf-1', + getPtyId: () => 'pty-1', + getSettings: () => null, + inspectProcess, + dispatchCompletion: vi.fn(), + isLive: () => true, + shouldPollProcessCadence: () => true + }) + coordinator.startProcessTracking() + coordinator.observeTitle('Codex working') + vi.advanceTimersByTime(3_000) + await flushAsyncTicks() + expect(inspectProcess).toHaveBeenCalled() + for (const call of inspectProcess.mock.calls) { + expect(optionsOf(call as unknown[])).toEqual({ steadyState: true }) + } + coordinator.dispose() + }) + + it('a pending-title read on a local pane is NOT steadyState: it decides a completion once', async () => { + const inspectProcess = vi.fn(async () => processResult('codex')) + const coordinator = createAgentCompletionCoordinator({ + paneKey: 'tab-1:leaf-1', + getPtyId: () => 'pty-1', + getSettings: () => null, + inspectProcess, + dispatchCompletion: vi.fn(), + isLive: () => true, + shouldPollProcessCadence: () => false + }) + coordinator.observeTitle('Codex working') + coordinator.observeTitle('/tmp/orca-e2e-repo') + await flushAsyncTicks() + expect(inspectProcess).toHaveBeenCalled() + for (const call of inspectProcess.mock.calls) { + expect(optionsOf(call as unknown[])).toBeUndefined() + } + coordinator.dispose() + }) + + it('never marks a remote pane as steadyState: remote identity needs evidence', async () => { + const inspectProcess = vi.fn(async () => processResult('codex')) + const coordinator = createAgentCompletionCoordinator({ + paneKey: 'tab-1:leaf-1', + getPtyId: () => 'remote:pty-1', + isRemotePtyId: () => true, + getExpectedIncarnationId: () => 'inc-1', + getSettings: () => null, + inspectProcess, + dispatchCompletion: vi.fn(), + isLive: () => true, + shouldPollProcessCadence: () => true + }) + coordinator.startProcessTracking() + coordinator.observeTitle('Codex working') + coordinator.observeTitle('/tmp/orca-e2e-repo') + vi.advanceTimersByTime(3_000) + await flushAsyncTicks() + expect(inspectProcess).toHaveBeenCalled() + for (const call of inspectProcess.mock.calls) { + expect(optionsOf(call as unknown[])).toEqual({ expectedIncarnationId: 'inc-1' }) + } + coordinator.dispose() + }) +}) diff --git a/src/renderer/src/runtime/runtime-terminal-inspection.ts b/src/renderer/src/runtime/runtime-terminal-inspection.ts index 8c354cdc415..b34d75f4555 100644 --- a/src/renderer/src/runtime/runtime-terminal-inspection.ts +++ b/src/renderer/src/runtime/runtime-terminal-inspection.ts @@ -138,7 +138,7 @@ export function recordRuntimeTerminalInputForPtyId(ptyId: string, timestamp = Da export async function inspectRuntimeTerminalProcess( settings: Pick | null | undefined, ptyId: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean; steadyState?: boolean } ): Promise { const ownerEnvironmentId = getRemoteRuntimePtyEnvironmentId(ptyId) const target = ownerEnvironmentId diff --git a/src/shared/cheap-process-table-snapshot-reader.ts b/src/shared/cheap-process-table-snapshot-reader.ts new file mode 100644 index 00000000000..1d2401f2045 --- /dev/null +++ b/src/shared/cheap-process-table-snapshot-reader.ts @@ -0,0 +1,51 @@ +import { runProcess } from './child-process/run-process' +import { + CHEAP_PS_ARGS, + PS_MAX_BUFFER_BYTES, + ProcessTableCaptureError, + parseCheapProcessTableRows, + type CheapProcessTableRow +} from './process-table-snapshot' +import { + PS_TIMEOUT_MS, + createProcessTableSnapshotReader, + withEvidenceBudget +} from './process-table-snapshot-reader' + +/** + * The cheap-tier sibling of the strict evidence reader: same coalescing and TTL, a + * column set without `tty=`/`command=`. Separate instance because the two column sets + * parse differently and a cheap capture must never be served to an evidence consumer. + */ +const cheapProcessTableReader = createProcessTableSnapshotReader({ + runPs: async () => { + const result = await runProcess({ + program: 'ps', + args: CHEAP_PS_ARGS, + timeoutMs: PS_TIMEOUT_MS, + maxOutputBytes: PS_MAX_BUFFER_BYTES + }) + // A ceiling hit is truncation, not absence: name it in the domain vocabulary. + if (result.outputTruncated) { + throw new ProcessTableCaptureError('capture_truncated') + } + if (result.timedOut) { + throw new ProcessTableCaptureError('capture_timeout') + } + if (result.code !== 0) { + throw new ProcessTableCaptureError(`ps_exit_${result.code ?? result.signal ?? 'unknown'}`) + } + return parseCheapProcessTableRows(result.stdout) + }, + now: () => Date.now() +}) + +/** Same wait bound as the evidence read: a stalled cheap capture must fall through to the full + * path's own handling rather than pin a polled tick. */ +export async function getCheapProcessTableSnapshot(): Promise { + return withEvidenceBudget(cheapProcessTableReader.getSnapshot()) +} + +export function resetCheapProcessTableSnapshotForTests(): void { + cheapProcessTableReader.reset() +} diff --git a/src/shared/cheap-process-table-snapshot.test.ts b/src/shared/cheap-process-table-snapshot.test.ts new file mode 100644 index 00000000000..74bbba2b69b --- /dev/null +++ b/src/shared/cheap-process-table-snapshot.test.ts @@ -0,0 +1,147 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { runProcessMock } = vi.hoisted(() => ({ runProcessMock: vi.fn() })) + +// The cheap reader goes through Orca's single child-process entry point (windowsHide, argv +// encoding, tree termination); mock at that seam rather than node:child_process. +vi.mock('./child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { + getCheapProcessTableSnapshot, + resetCheapProcessTableSnapshotForTests +} from './cheap-process-table-snapshot-reader' +import { PS_TIMEOUT_MS } from './process-table-snapshot-reader' +import { + CHEAP_PS_ARGS, + PS_ARGS, + PS_MAX_BUFFER_BYTES, + parseCheapProcessTableRows, + ProcessTableCaptureError +} from './process-table-snapshot' + +function installPs(stdout: string, outputTruncated = false): string[][] { + const calls: string[][] = [] + runProcessMock.mockImplementation(async (spec: { program: string; args: readonly string[] }) => { + calls.push([spec.program, ...spec.args]) + return { code: 0, signal: null, stdout, stderr: '', timedOut: false, outputTruncated } + }) + return calls +} + +describe('parseCheapProcessTableRows', () => { + it('parses the macOS column set with a padded lstart marker', () => { + const rows = parseCheapProcessTableRows( + [ + ' 1 0 1 0 Ss Tue Sep 1 01:49:39 2026', + ' 4242 4200 4242 4243 S Thu Sep 3 16:02:01 2026', + ' 4243 4242 4243 4243 S+ Thu Sep 3 16:02:05 2026', + '' + ].join('\n') + ) + expect(rows).toEqual([ + { pid: 1, ppid: 0, pgid: 1, tpgid: 0, stat: 'Ss', startTime: 'Tue Sep 1 01:49:39 2026' }, + { + pid: 4242, + ppid: 4200, + pgid: 4242, + tpgid: 4243, + stat: 'S', + startTime: 'Thu Sep 3 16:02:01 2026' + }, + { + pid: 4243, + ppid: 4242, + pgid: 4243, + tpgid: 4243, + stat: 'S+', + startTime: 'Thu Sep 3 16:02:05 2026' + } + ]) + }) + + it('parses the Linux column set, which carries no start marker', () => { + const rows = parseCheapProcessTableRows( + ' 2 0 0 -1 S\r\n 900 1 900 900 Ss+\r\n' + ) + expect(rows).toEqual([ + { pid: 2, ppid: 0, pgid: 0, tpgid: -1, stat: 'S' }, + { pid: 900, ppid: 1, pgid: 900, tpgid: 900, stat: 'Ss+' } + ]) + }) + + it('skips malformed rows rather than failing the capture', () => { + expect(parseCheapProcessTableRows('garbage\n 7 1 7 7 S\n')).toEqual([ + { pid: 7, ppid: 1, pgid: 7, tpgid: 7, stat: 'S' } + ]) + }) + + it('treats an empty capture as unreadable, never as "no processes"', () => { + expect(() => parseCheapProcessTableRows('\n\n')).toThrow(ProcessTableCaptureError) + }) +}) + +describe('getCheapProcessTableSnapshot', () => { + beforeEach(() => { + runProcessMock.mockReset() + resetCheapProcessTableSnapshotForTests() + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(0) + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('forks ps with the cheap column set only, never tty or command', async () => { + const calls = installPs(' 7 1 7 7 S\n') + await getCheapProcessTableSnapshot() + expect(calls).toEqual([['ps', ...CHEAP_PS_ARGS]]) + expect(CHEAP_PS_ARGS.join(' ')).not.toMatch(/tty=|command=|etimes=/) + expect(CHEAP_PS_ARGS).not.toEqual(PS_ARGS) + }) + + it('coalesces concurrent readers onto one fork and honours the TTL', async () => { + const calls = installPs(' 7 1 7 7 S\n') + await Promise.all([getCheapProcessTableSnapshot(), getCheapProcessTableSnapshot()]) + await getCheapProcessTableSnapshot() + expect(calls).toHaveLength(1) + vi.setSystemTime(600) + await getCheapProcessTableSnapshot() + expect(calls).toHaveLength(2) + }) + + it('passes the full-tier buffer ceiling and timeout to the runner', async () => { + installPs(' 7 1 7 7 S\n') + await getCheapProcessTableSnapshot() + expect(runProcessMock).toHaveBeenCalledWith( + expect.objectContaining({ maxOutputBytes: PS_MAX_BUFFER_BYTES, timeoutMs: PS_TIMEOUT_MS }) + ) + }) + + it('names a clipped capture as truncated, a killed one as a timeout, and a non-zero exit by its code', async () => { + installPs(' 7 1 7 7 S\n', true) + await expect(getCheapProcessTableSnapshot()).rejects.toMatchObject({ + reason: 'capture_truncated' + }) + resetCheapProcessTableSnapshotForTests() + runProcessMock.mockResolvedValueOnce({ + code: null, + signal: 'SIGKILL', + stdout: '', + stderr: '', + timedOut: true + }) + await expect(getCheapProcessTableSnapshot()).rejects.toMatchObject({ + reason: 'capture_timeout' + }) + resetCheapProcessTableSnapshotForTests() + runProcessMock.mockResolvedValueOnce({ + code: 1, + signal: null, + stdout: '', + stderr: 'ps: bad column', + timedOut: false + }) + await expect(getCheapProcessTableSnapshot()).rejects.toMatchObject({ reason: 'ps_exit_1' }) + }) +}) diff --git a/src/shared/process-table-snapshot-reader.ts b/src/shared/process-table-snapshot-reader.ts index a40b934c0e1..b24f2a1f2dd 100644 --- a/src/shared/process-table-snapshot-reader.ts +++ b/src/shared/process-table-snapshot-reader.ts @@ -207,6 +207,19 @@ function assertWholeCapture(stdout: string): string { return stdout } +/** Field 22 (`starttime`) of `/proc//stat`, read past the parenthesised comm. */ +export function parseLinuxProcStatStartTime(stat: string): string | null { + const closingParen = stat.lastIndexOf(')') + if (closingParen === -1) { + return null + } + const tail = stat + .slice(closingParen + 1) + .trim() + .split(/\s+/) + return tail[19] || null +} + /** Read Linux's stable PID start-time ticks without spawning another process. */ async function readLinuxProcessStartTimes( rows: readonly ProcessTableRow[] @@ -218,16 +231,9 @@ async function readLinuxProcessStartTimes( const starts = await Promise.all( candidates.map(async (row) => { try { - const stat = await readFile(`/proc/${row.pid}/stat`, 'utf8') - const closingParen = stat.lastIndexOf(')') - if (closingParen === -1) { - return null - } - const tail = stat - .slice(closingParen + 1) - .trim() - .split(/\s+/) - const startTime = tail[19] + const startTime = parseLinuxProcStatStartTime( + await readFile(`/proc/${row.pid}/stat`, 'utf8') + ) return startTime ? ([row.pid, startTime] as const) : null } catch { return null @@ -300,7 +306,7 @@ export const PROCESS_TABLE_EVIDENCE_BUDGET_MS = 1_200 * capture some identity probe started under the 15s budget; abandoning the wait leaves that * capture running to fill the cache instead of forking a second whole-machine `ps` on the host * that can least afford one. */ -async function withEvidenceBudget(pending: Promise): Promise { +export async function withEvidenceBudget(pending: Promise): Promise { let timer: ReturnType | undefined try { return await Promise.race([ diff --git a/src/shared/process-table-snapshot.ts b/src/shared/process-table-snapshot.ts index 00cf1f1c380..0acc6f6f9e9 100644 --- a/src/shared/process-table-snapshot.ts +++ b/src/shared/process-table-snapshot.ts @@ -20,6 +20,60 @@ export const PS_ARGS = ( : ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat=,tty=,etimes=,command='] ) as readonly string[] +/** + * Cheap tier: the same job-control columns without `tty=` (0.29s of the 0.34s on a + * 1,900-process Mac) or `command=` (per-pid argv read, 1.15s on Linux). Enough to prove a + * pane's subtree is unchanged since the last full capture; never enough to name a process. + * No `etimes=` on Linux: it is elapsed seconds, so it changes every tick; the stable start + * marker comes from `/proc//stat` for the pane subtree only. + */ +export const CHEAP_PS_ARGS = ( + process.platform === 'darwin' + ? ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat=,lstart='] + : ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat='] +) as readonly string[] + +export type CheapProcessTableRow = { + pid: number + ppid: number + pgid: number + tpgid: number + stat: string + /** Host start marker when the column set carries one (macOS `lstart`). */ + startTime?: string +} + +/** + * Parse a {@link CHEAP_PS_ARGS} capture. Lenient on purpose: a dropped row can only make a + * fingerprint DIFFER from the strict full-capture one, which escalates to the full capture -- + * the safe direction. An empty capture is unreadable, not "no processes". + */ +export function parseCheapProcessTableRows(stdout: string): CheapProcessTableRow[] { + const rows: CheapProcessTableRow[] = [] + for (const rawLine of stdout.split(/\r?\n/)) { + const match = rawLine.trim().match(/^(\d+)\s+(\d+)\s+(-?\d+)\s+(-?\d+)\s+(\S+)(?:\s+(.+?))?$/) + if (!match) { + continue + } + const pid = Number(match[1]) + if (!Number.isSafeInteger(pid) || pid <= 0) { + continue + } + rows.push({ + pid, + ppid: Number(match[2]), + pgid: Number(match[3]), + tpgid: Number(match[4]), + stat: match[5], + ...(match[6] !== undefined ? { startTime: match[6] } : {}) + }) + } + if (rows.length === 0) { + throw new ProcessTableCaptureError('empty_capture') + } + return rows +} + // Why: execFile's 1MB default leaves ~3x headroom (326KB / 1,460 processes, and // a single 5KB argv row is ordinary), so a busy host overflows it and then EVERY // capture fails — a readable process table degrading into permanent