diff --git a/config/scripts/build-mobile-web-app-bundle.test.mjs b/config/scripts/build-mobile-web-app-bundle.test.mjs index baa18b6b39d..489599d7bf3 100644 --- a/config/scripts/build-mobile-web-app-bundle.test.mjs +++ b/config/scripts/build-mobile-web-app-bundle.test.mjs @@ -539,10 +539,10 @@ describe('the Phase C budget', () => { key ).toBeGreaterThanOrEqual(measured) } - // 1 to 9 per route, which is why four per route was a bound rather than a fit and why the + // 1 to 10 per route, which is why four per route was a bound rather than a fit and why the // envelope cannot be a line through the measurement either. expect(Math.min(...MOBILE_WEB_APP_BUNDLE_ROUTE_SCRIPT_SPREAD)).toBe(1) - expect(Math.max(...MOBILE_WEB_APP_BUNDLE_ROUTE_SCRIPT_SPREAD)).toBe(9) + expect(Math.max(...MOBILE_WEB_APP_BUNDLE_ROUTE_SCRIPT_SPREAD)).toBe(10) }) it('sits exactly one margin over the swept tree and grants the worst route beyond it', () => { @@ -614,13 +614,13 @@ describe('the Phase C budget', () => { it('fails the build when the derived ceiling passes what the phone will accept', async () => { // The shell hands back null for a manifest over its own ceiling, so a derived ceiling above // that ships a green build no device can open. At the 42 images the tree carries, the envelope - // plus 42 plus the document crosses 256 at 32 routes, which Phase C reaches. The crossing came + // plus 42 plus the document crosses 256 at 30 routes, which Phase C reaches. The crossing came // in from 50 with the envelope: it grants the worst swept route to each one past the sweep, // where `4r + 16` granted four, so re-measuring a tree whose routes share more moves it out. expect(await readMobileWebBundleMaxAssets()).toBe(MOBILE_WEB_BUNDLE_MAX_ASSETS) - expect(assertAssetCeilingFitsShell(31, 42, MOBILE_WEB_BUNDLE_MAX_ASSETS)).toBe(251) - expect(() => assertAssetCeilingFitsShell(32, 42, MOBILE_WEB_BUNDLE_MAX_ASSETS)).toThrow( - /260 .*256/ + expect(assertAssetCeilingFitsShell(29, 42, MOBILE_WEB_BUNDLE_MAX_ASSETS)).toBe(251) + expect(() => assertAssetCeilingFitsShell(30, 42, MOBILE_WEB_BUNDLE_MAX_ASSETS)).toThrow( + /261 .*256/ ) }) }) diff --git a/config/scripts/verify-mobile-web-app-bundle.mjs b/config/scripts/verify-mobile-web-app-bundle.mjs index b6297555701..48917d294ce 100644 --- a/config/scripts/verify-mobile-web-app-bundle.mjs +++ b/config/scripts/verify-mobile-web-app-bundle.mjs @@ -50,9 +50,9 @@ export const MOBILE_WEB_APP_BUNDLE_MAX_TOTAL_BYTES = 9 * 1024 * 1024 * A chunk is emitted per distinct set of importers, not per route, so a route's marginal cost is * what it fails to share rather than what it weighs. Re-measured on this head by building * `routes.slice(0, n)` for every n, which is what the fence below is derived from rather than - * fitted to. The spread it shows is 1 to 9: `pr` and `web` add one script each, `review` adds nine. + * fitted to. The spread it shows is 1 to 10: `pr` and `web` add one script each, `session` adds ten. * The root `./_layout.tsx` (the page's web sibling of the native root) sorts first; with it the - * swept tree reads 69 scripts at 16 routes, the old 15 read 67 on the same head. + * swept tree reads 74 scripts at 16 routes. * * This table is the fence's only input, so a route added to the tree stales it and the pins beside * the fence fail until it is re-measured. That is the point: the bound is re-derived, never bumped. @@ -61,19 +61,19 @@ export const MOBILE_WEB_APP_BUNDLE_SCRIPT_SWEEP = [ ['./_layout.tsx', 3], ['./h/[hostId]/[...page].tsx', 7], ['./h/[hostId]/accounts.tsx', 9], - ['./h/[hostId]/agent-history/[worktreeId].tsx', 13], - ['./h/[hostId]/edit.tsx', 18], - ['./h/[hostId]/files/[worktreeId].tsx', 21], - ['./h/[hostId]/files/preview/[worktreeId].tsx', 28], - ['./h/[hostId]/history/[worktreeId].tsx', 30], - ['./h/[hostId]/index.tsx', 35], - ['./h/[hostId]/pr/[worktreeId].tsx', 36], - ['./h/[hostId]/review/[worktreeId].tsx', 45], - ['./h/[hostId]/session/[worktreeId].tsx', 54], - ['./h/[hostId]/source-control/[worktreeId].tsx', 59], - ['./h/[hostId]/tasks.tsx', 66], - ['./h/[hostId]/web.tsx', 67], - ['./h/_layout.tsx', 69] + ['./h/[hostId]/agent-history/[worktreeId].tsx', 14], + ['./h/[hostId]/edit.tsx', 19], + ['./h/[hostId]/files/[worktreeId].tsx', 24], + ['./h/[hostId]/files/preview/[worktreeId].tsx', 32], + ['./h/[hostId]/history/[worktreeId].tsx', 34], + ['./h/[hostId]/index.tsx', 39], + ['./h/[hostId]/pr/[worktreeId].tsx', 40], + ['./h/[hostId]/review/[worktreeId].tsx', 48], + ['./h/[hostId]/session/[worktreeId].tsx', 58], + ['./h/[hostId]/source-control/[worktreeId].tsx', 63], + ['./h/[hostId]/tasks.tsx', 71], + ['./h/[hostId]/web.tsx', 72], + ['./h/_layout.tsx', 74] ] const sweptScripts = MOBILE_WEB_APP_BUNDLE_SCRIPT_SWEEP.map(([, scripts]) => scripts) @@ -191,7 +191,7 @@ export async function readMobileWebBundleMaxAssets() { * shells return null for a manifest over MOBILE_WEB_BUNDLE_MAX_ASSETS rather than dropping the * extra assets, so a route count that pushes the chunk envelope plus images plus the document past * it would pass this build and fail on the device with nothing to read. At today's 42 images that - * is 31 routes, inside what Phase C adds, which is why this is a build failure and not a comment. + * is 29 routes, inside what Phase C adds, which is why this is a build failure and not a comment. * The envelope grants the worst swept route to each one past the sweep, so re-measuring a tree * whose routes share more moves that crossing out again. */ diff --git a/mobile/src/session/mobile-existing-agent-launch.ts b/mobile/src/session/mobile-existing-agent-launch.ts index 7cfb5cc2e47..3531838291b 100644 --- a/mobile/src/session/mobile-existing-agent-launch.ts +++ b/mobile/src/session/mobile-existing-agent-launch.ts @@ -7,19 +7,17 @@ */ import type { AgentLaunchPrompt, AgentLaunchResult } from '../../../src/shared/agent-launch-intent' -import { AGENT_LAUNCH_PANE_ALREADY_LIVE_CODE } from '../../../src/shared/agent-launch-pane-already-live' -import { AGENT_LAUNCH_SESSION_ALREADY_EXISTS_CODE } from '../../../src/shared/agent-launch-session-already-exists' import { isAgentSessionHandleProvider } from '../../../src/shared/agent-session-provider-handle' import { makePaneKey } from '../../../src/shared/stable-pane-id' import { createStructuredAgentSessionId } from '../../../src/shared/structured-agent-session-create' import type { TuiAgent } from '../../../src/shared/tui-agent' import type { RpcClient } from '../transport/rpc-client' import { agentLaunchReplayRun } from '../tasks/mobile-workspace-create-operations' +import { agentLaunchExistingParams, readAgentLaunchSupport } from '../tasks/agent-launch-request' import { - agentLaunchExistingParams, - isAgentLaunchReplayUnsupportedRefusal, - readAgentLaunchSupport -} from '../tasks/agent-launch-request' + classifyAgentLaunchReplayRefusal, + isAgentLaunchReservationTakenRefusal +} from '../../../src/shared/agent-launch-replay-refusal' import { sendReplayingAmbiguousDelivery } from '../tasks/replay-on-ambiguous-delivery' import { structuredSessionOperationId, @@ -152,29 +150,17 @@ function classifyLaunchRefusal( error: { code?: string; message?: string }, replayed: boolean ): MobileExistingAgentLaunch { - if (isAgentLaunchReplayUnsupportedRefusal(error)) { - // Only a refusal of the first send proves nothing ran; after a replay it may be a replacement - // connection whose capability list hasn't landed, answering for an attempt that did start. - return replayed - ? { kind: 'unknown', message: AGENT_LAUNCH_UNCONFIRMED_MESSAGE } - : { kind: 'unsupported' } + switch (classifyAgentLaunchReplayRefusal(error, replayed)) { + case 'unsupported': + return { kind: 'unsupported' } + case 'unknown': + return { kind: 'unknown', message: AGENT_LAUNCH_UNCONFIRMED_MESSAGE } + case 'failed': { + if (isAgentLaunchReservationTakenRefusal(error)) { + return { kind: 'failed', message: AGENT_LAUNCH_RESERVATION_TAKEN_MESSAGE } + } + const message = error.message?.trim() + return { kind: 'failed', message: message || "Couldn't start the agent." } + } } - if ( - error.code === 'agent_session_operation_unknown' || - error.code === 'agent_session_operation_expired' - ) { - return { kind: 'unknown', message: AGENT_LAUNCH_UNCONFIRMED_MESSAGE } - } - if ( - error.code === AGENT_LAUNCH_PANE_ALREADY_LIVE_CODE || - error.code === AGENT_LAUNCH_SESSION_ALREADY_EXISTS_CODE - ) { - // A taken reservation proves nothing started only on the first send; after a replay the pane - // or chat holding it may be this launch's own. - return replayed - ? { kind: 'unknown', message: AGENT_LAUNCH_UNCONFIRMED_MESSAGE } - : { kind: 'failed', message: AGENT_LAUNCH_RESERVATION_TAKEN_MESSAGE } - } - const message = error.message?.trim() - return { kind: 'failed', message: message || "Couldn't start the agent." } } diff --git a/mobile/src/tasks/agent-launch-mobile-replay.test.ts b/mobile/src/tasks/agent-launch-mobile-replay.test.ts index c68fb4ff7c1..515cf353f3b 100644 --- a/mobile/src/tasks/agent-launch-mobile-replay.test.ts +++ b/mobile/src/tasks/agent-launch-mobile-replay.test.ts @@ -20,7 +20,7 @@ import { runtimeStub } from '../../../src/main/runtime/rpc/methods/agent-launch. import { AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY, AGENT_LAUNCH_RUNTIME_CAPABILITY -} from '../../../src/shared/protocol-version' +} from '../../../src/shared/agent-launch-runtime-capability' import { launchAgentInExistingWorkspace, reserveMobileAgentLaunch diff --git a/mobile/src/tasks/agent-launch-request.ts b/mobile/src/tasks/agent-launch-request.ts index 0c3c71687f0..b6d730d6e2e 100644 --- a/mobile/src/tasks/agent-launch-request.ts +++ b/mobile/src/tasks/agent-launch-request.ts @@ -24,7 +24,7 @@ import { import { AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY, AGENT_LAUNCH_RUNTIME_CAPABILITY -} from '../../../src/shared/protocol-version' +} from '../../../src/shared/agent-launch-runtime-capability' import type { TuiAgent } from '../../../src/shared/tui-agent' import type { RpcSendParams } from '../transport/rpc-params-contract' import type { WorkspaceCreateParams } from './workspace-create-params' @@ -142,11 +142,5 @@ export function isAgentLaunchUnsupportedRefusal(error: { return (error.message ?? '').includes('agent_launch_unsupported') } -/** The `agent.launchReplay` twin: an older host rejects the method rather than a field. */ -export function isAgentLaunchReplayUnsupportedRefusal(error: { code?: string }): boolean { - return ( - error.code === 'method_not_found' || - error.code === 'forbidden' || - error.code === 'agent_launch_replay_unsupported' - ) -} +// The `agent.launchReplay` twin lives in shared so the desktop classifies refusals the same way. +export { isAgentLaunchReplayUnsupportedRefusal } from '../../../src/shared/agent-launch-replay-refusal' diff --git a/mobile/src/tasks/mobile-agent-launch-architecture.test.ts b/mobile/src/tasks/mobile-agent-launch-architecture.test.ts index 3fc2ce2a4ca..60f09c83a18 100644 --- a/mobile/src/tasks/mobile-agent-launch-architecture.test.ts +++ b/mobile/src/tasks/mobile-agent-launch-architecture.test.ts @@ -23,7 +23,7 @@ import { readNewWorktreeRuntimeCapabilities } from './worktree-create-capability import { AGENT_LAUNCH_RUNTIME_CAPABILITY, AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY -} from '../../../src/shared/protocol-version' +} from '../../../src/shared/agent-launch-runtime-capability' const createStructuredSession = vi.fn() vi.mock('../../../src/main/runtime/rpc/methods/structured-agent-session-create', () => ({ diff --git a/mobile/src/tasks/worktree-create-capability.test.ts b/mobile/src/tasks/worktree-create-capability.test.ts index 830e1a61805..78fa42be9f8 100644 --- a/mobile/src/tasks/worktree-create-capability.test.ts +++ b/mobile/src/tasks/worktree-create-capability.test.ts @@ -3,7 +3,7 @@ import { AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY, AGENT_LAUNCH_REPLAY_RUNTIME_CAPABILITY, AGENT_LAUNCH_RUNTIME_CAPABILITY -} from '../../../src/shared/protocol-version' +} from '../../../src/shared/agent-launch-runtime-capability' import type { RpcClient } from '../transport/rpc-client' import { LogicalClientCutoverError } from '../transport/stable-logical-rpc-client' import { readNewWorktreeRuntimeCapabilities } from './worktree-create-capability' diff --git a/mobile/src/transport/mobile-runtime-client-capabilities.test.ts b/mobile/src/transport/mobile-runtime-client-capabilities.test.ts index b0c38166dff..45502e1f028 100644 --- a/mobile/src/transport/mobile-runtime-client-capabilities.test.ts +++ b/mobile/src/transport/mobile-runtime-client-capabilities.test.ts @@ -1,12 +1,12 @@ import { describe, expect, it } from 'vitest' import { - AGENT_LAUNCH_RUNTIME_CAPABILITY, AGENT_SESSION_TURN_ITEM_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, SESSION_TABS_SPLIT_GROUP_PLACEMENT_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../src/shared/protocol-version' +import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../../src/shared/agent-launch-runtime-capability' import { MOBILE_RUNTIME_CLIENT_CAPABILITIES } from './mobile-runtime-client-capabilities' /** Mirrors the host's `parseRuntimeClientCapabilities`, which returns an EMPTY list — silently diff --git a/mobile/src/transport/mobile-runtime-client-capabilities.ts b/mobile/src/transport/mobile-runtime-client-capabilities.ts index e388c7ff2d8..32dc88a6280 100644 --- a/mobile/src/transport/mobile-runtime-client-capabilities.ts +++ b/mobile/src/transport/mobile-runtime-client-capabilities.ts @@ -1,5 +1,4 @@ import { - AGENT_LAUNCH_RUNTIME_CAPABILITY, AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, AGENT_SESSION_TURN_ITEM_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, @@ -7,6 +6,7 @@ import { STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../src/shared/protocol-version' +import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../../src/shared/agent-launch-runtime-capability' import { remoteRuntimeClientCapabilities } from '../../../src/shared/remote-runtime-client-capabilities' export const MOBILE_RUNTIME_CLIENT_CAPABILITIES = remoteRuntimeClientCapabilities([ diff --git a/src/main/agent-launch/__fixtures__/host-agent-startup-call-sites.txt b/src/main/agent-launch/__fixtures__/host-agent-startup-call-sites.txt index 193c69c2765..b7cff65366c 100644 --- a/src/main/agent-launch/__fixtures__/host-agent-startup-call-sites.txt +++ b/src/main/agent-launch/__fixtures__/host-agent-startup-call-sites.txt @@ -2,8 +2,8 @@ # attributes a fresh agent the host builds; carries agentStartedTelemetry # resume continues an existing agent session; not a new start # mobile-followup the phone's session-tab launch; attribution is a separate follow-up -# The shared execution-host builder also serves bare commands; only fresh-agent callers supply attribution. -src/main/opencode/opencode-model-startup-plan.ts 2 attributes -src/main/runtime/runtime-worktree-agent-startup.ts 3 attributes +# The shared execution-host builders also serve bare commands; only fresh-agent callers supply attribution. +src/main/opencode/opencode-model-startup-plan.ts 3 attributes +src/main/runtime/runtime-worktree-agent-startup.ts 4 attributes src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts 1 resume src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts 1 mobile-followup diff --git a/src/main/agent-launch/agent-launch-executor.test.ts b/src/main/agent-launch/agent-launch-executor.test.ts index e7270287fb5..210289785ac 100644 --- a/src/main/agent-launch/agent-launch-executor.test.ts +++ b/src/main/agent-launch/agent-launch-executor.test.ts @@ -27,8 +27,12 @@ function harness(options: { structuredCreateError?: Error deliveredMessageId?: string | null terminalPromptDelivered?: boolean + /** Whether the surface reports that its typed line took the offered prompt. */ + lineCarriesPrompt?: boolean }) { const calls: string[] = [] + const carried = (startupPrompt: string | undefined) => + startupPrompt && (options.lineCarriesPrompt ?? true) ? { promptRodeLaunchCommand: true } : {} const createWorktree = vi.fn( async (args: { create: Record @@ -38,7 +42,8 @@ function harness(options: { calls.push(`createWorktree(startupAgent=${String(args.startupAgent)})`) return { worktreeId: 'wt-new', - startupTerminalHandle: args.startupAgent ? 'term_agent_first' : undefined + startupTerminalHandle: args.startupAgent ? 'term_agent_first' : undefined, + ...carried(args.startupPrompt) } } ) @@ -56,9 +61,9 @@ function harness(options: { } return { sessionId: 'sess-1', handle: 'handle_structured', fence: 4 } }) - const createTerminalAgent = vi.fn(async (_args: { startupPrompt?: string }) => { + const createTerminalAgent = vi.fn(async (args: { startupPrompt?: string }) => { calls.push('createTerminalAgent') - return { handle: 'term_1' } + return { handle: 'term_1', ...carried(args.startupPrompt) } }) const deliverStructuredPrompt = vi.fn(async () => { calls.push('deliverStructuredPrompt') @@ -297,10 +302,10 @@ describe('the prompt receipt', () => { }) /** - * A terminal takes its prompt one of two ways, and which one is not a preference: an agent whose - * CLI accepts a prompt argument must get it on argv, because that is the transport that survives - * multi-line and special-character text. Only an agent with no such argument is written to as - * keystrokes. `claude` is argv-mode, `aider` is `stdin-after-start` — the two halves of the table. + * A terminal takes its prompt one of two ways. An agent whose CLI accepts a prompt argument is + * offered it on the launch command, and the surface that types that line reports whether it rode; + * everything else is written as keystrokes once the agent is ready. `claude` is argv-mode, `aider` + * is `stdin-after-start` — the two halves of the table. */ describe('delivering a launch prompt to a terminal agent', () => { const SUBMIT = { text: 'do the thing', delivery: 'submit' } as const @@ -329,6 +334,8 @@ describe('delivering a launch prompt to a terminal agent', () => { expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) expect(h.deliverTerminalPrompt).toHaveBeenCalledWith({ handle: 'term_1', + agent: 'aider', + freshLaunch: true, prompt: SUBMIT }) // Folding it into argv would have appended it as an argument the CLI does not accept. @@ -348,6 +355,39 @@ describe('delivering a launch prompt to a terminal agent', () => { expect(h.deliverTerminalPrompt).not.toHaveBeenCalled() }) + it('pastes an argv agent’s prompt after start when the surface reports its typed line could not carry it', async () => { + const h = harness({ + createSupport: { supported: false, reason: 'wsl' }, + lineCarriesPrompt: false + }) + const result = await h.run({ ...CREATE_INTENT, prompt: SUBMIT }) + + expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) + // Offered to the launch command; the surface, not the executor, decided it did not ride. + expect(h.createTerminalAgent.mock.calls[0]?.[0]).toMatchObject({ + startupPrompt: 'do the thing' + }) + expect(h.deliverTerminalPrompt).toHaveBeenCalledWith({ + handle: 'term_1', + agent: 'claude', + freshLaunch: true, + prompt: SUBMIT + }) + }) + + it('pastes into an agent-first create’s startup terminal when its typed line could not carry the prompt', async () => { + const h = harness({ settings: null, lineCarriesPrompt: false }) + const result = await h.run({ ...CREATE_INTENT, prompt: SUBMIT }) + + expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) + expect(h.deliverTerminalPrompt).toHaveBeenCalledWith({ + handle: 'term_agent_first', + agent: 'claude', + freshLaunch: true, + prompt: SUBMIT + }) + }) + it('writes into a reused terminal, whose process started before the launch existed', async () => { const h = harness({}) const result = await h.run({ @@ -361,6 +401,8 @@ describe('delivering a launch prompt to a terminal agent', () => { expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) expect(h.deliverTerminalPrompt).toHaveBeenCalledWith({ handle: 'term_existing', + agent: 'claude', + freshLaunch: false, prompt: SUBMIT }) }) diff --git a/src/main/agent-launch/agent-launch-executor.ts b/src/main/agent-launch/agent-launch-executor.ts index ac1511d1258..b33548abee0 100644 --- a/src/main/agent-launch/agent-launch-executor.ts +++ b/src/main/agent-launch/agent-launch-executor.ts @@ -71,8 +71,12 @@ export type AgentLaunchExecution = { vocabulary?: AgentLaunchModeVocabulary /** Attributes a throw to the step that was running, the way a dispatch's own stages do. */ onStage?: (stage: 'worktree_create' | 'mode_settle' | 'surface_create') => void + /** The surface exists and its tab is published; runs before any prompt delivery. Must not throw. */ + onSurfacePublished?: (surface: AgentLaunchPublishedSurface) => void } +export type AgentLaunchPublishedSurface = Pick + export async function executeAgentLaunch( execution: AgentLaunchExecution ): Promise { @@ -99,13 +103,18 @@ export async function executeAgentLaunch( // A reused terminal already downgraded in the pre-flight; there is nothing to create. Its agent // was running before this launch existed, so argv is unreachable and the PTY is the only way in. if (intent.reuseTerminal) { - return { + const reused = published(execution, { outcome: { kind: 'terminal', handle: intent.reuseTerminal.handle }, - worktreeId: existingWorktreeId(intent.target), + worktreeId: existingWorktreeId(intent.target) + }) + return { + ...reused, receipt: preflight, ...promptReceipt( intent, - await deliverTerminalLaunchPrompt(execution, intent.reuseTerminal.handle) + await deliverTerminalLaunchPrompt(execution, intent.reuseTerminal.handle, { + freshLaunch: false + }) ) } } @@ -113,20 +122,25 @@ export async function executeAgentLaunch( const placed = await resolveWorkspace(execution, preflight) // Agent-first creation already produced the agent, so the pre-flight verdict is final. if (placed.startupTerminalHandle) { - return { + const startup = published(execution, { outcome: { kind: 'terminal', handle: placed.startupTerminalHandle, ...(placed.startupTerminalPaneKey ? { paneKey: placed.startupTerminalPaneKey } : {}) }, - worktreeId: placed.worktreeId, + worktreeId: placed.worktreeId + }) + return { + ...startup, receipt: preflight, ...(placed.warning ? { warning: placed.warning } : {}), ...promptReceipt( intent, placed.promptRodeLaunchCommand ? HANDED_TO_TERMINAL - : await deliverTerminalLaunchPrompt(execution, placed.startupTerminalHandle) + : await deliverTerminalLaunchPrompt(execution, placed.startupTerminalHandle, { + freshLaunch: true + }) ) } } @@ -169,15 +183,23 @@ export async function executeAgentLaunch( // not start while looking at it. Telling those apart needs `createManagedWorktree` to stop // multiplexing "couldn't copy untracked files" and "startup terminal failed" into one string. const warning = combineLaunchWarnings(placed.warning, created.warning) + const surface = published(execution, { outcome: created.outcome, worktreeId: placed.worktreeId }) return { - outcome: created.outcome, - worktreeId: placed.worktreeId, + ...surface, receipt: settled, ...(warning ? { warning } : {}), ...promptReceipt(intent, await settleLaunchPromptDisposal(execution, created)) } } +function published( + execution: AgentLaunchExecution, + surface: AgentLaunchPublishedSurface +): AgentLaunchPublishedSurface { + execution.onSurfacePublished?.(surface) + return surface +} + function downgradeAgentLaunchModeForStructuredRefusal( receipt: AgentLaunchModeReceipt, vocabulary: AgentLaunchModeVocabulary @@ -223,9 +245,10 @@ async function resolveWorkspace( }) // Only when a startup terminal actually came back: a create that produced none ran no command, // so nothing carried the prompt and the launch still owes it to whatever surface it builds next. - return created.startupTerminalHandle && startupPrompt - ? { ...created, promptRodeLaunchCommand: true } - : created + const { promptRodeLaunchCommand, ...rest } = created + return rest.startupTerminalHandle && promptRodeLaunchCommand + ? { ...rest, promptRodeLaunchCommand: true } + : rest } /** `structured` is the same surface `outcome` names, kept typed so prompt delivery reads the create's @@ -327,7 +350,7 @@ async function createTerminalSurface( ...(terminal.paneKey ? { paneKey: terminal.paneKey } : {}) }, ...(terminal.warning ? { warning: terminal.warning } : {}), - ...(startupPrompt ? { promptRodeLaunchCommand: true } : {}) + ...(startupPrompt && terminal.promptRodeLaunchCommand ? { promptRodeLaunchCommand: true } : {}) } } diff --git a/src/main/agent-launch/agent-launch-prompt-delivery.ts b/src/main/agent-launch/agent-launch-prompt-delivery.ts index e77e0cabbe5..b3651901376 100644 --- a/src/main/agent-launch/agent-launch-prompt-delivery.ts +++ b/src/main/agent-launch/agent-launch-prompt-delivery.ts @@ -5,16 +5,15 @@ * surface exists and in what order; everything here decides how the text reaches whichever surface * that turned out to be, and each surface takes it differently: * - * structured session -> committed to the transcript, named by a message id -> journaled - * terminal, argv CLI -> folded into the command that execs the agent -> handed-to-terminal - * terminal, no argv -> bracketed paste into the live PTY -> handed-to-terminal - * anything unproven -> -> not-delivered + * structured session -> committed to the transcript, named by a message id -> journaled + * terminal, line fits -> folded into the command that execs the agent -> handed-to-terminal + * terminal, otherwise -> bracketed paste into the live PTY once it is ready -> handed-to-terminal + * anything unproven -> -> not-delivered * - * The argv/PTY fork is not a preference. `argv` exists so multi-line and special-character text - * reaches a CLI as one argument instead of keystrokes, and it has no readiness race because the - * text is in the process's arguments at exec time. So it is preferred wherever the agent's CLI - * takes a prompt argument, and `agentPromptRidesLaunchCommand` — derived from the same injection - * table `buildAgentStartupPlan` branches on — is the one place that question is asked. + * argv has no readiness race, so it is offered wherever the agent's CLI takes a prompt argument + * (`agentPromptRidesLaunchCommand`). But that command is TYPED into the user's shell, and a long or + * multi-line typed line fails in ways argv itself does not, so whether the offer was taken is decided + * where the line is built (`startup-line-prompt-carry`) and reported back, never predicted here. */ import type { @@ -39,11 +38,11 @@ export async function settleLaunchPromptDisposal( const messageId = await deliverStructuredLaunchPrompt(execution, created.structured) return messageId ? { outcome: 'journaled', messageId } : NOT_DELIVERED } - // The startup command already carries an argv agent's prompt; there is nothing left to write. + // The surface reported that the startup command carried the prompt; there is nothing left to write. if (created.promptRodeLaunchCommand) { return HANDED_TO_TERMINAL } - return deliverTerminalLaunchPrompt(execution, created.outcome.handle) + return deliverTerminalLaunchPrompt(execution, created.outcome.handle, { freshLaunch: true }) } /** @@ -82,13 +81,19 @@ async function deliverStructuredLaunchPrompt( */ export async function deliverTerminalLaunchPrompt( execution: AgentLaunchExecution, - handle: string + handle: string, + { freshLaunch }: { freshLaunch: boolean } ): Promise { const { intent, surfaces } = execution if (!intent.prompt || intent.prompt.delivery !== 'submit') { return NOT_DELIVERED } - const delivered = await surfaces.deliverTerminalPrompt?.({ handle, prompt: intent.prompt }) + const delivered = await surfaces.deliverTerminalPrompt?.({ + handle, + agent: intent.agent, + freshLaunch, + prompt: intent.prompt + }) return delivered ? HANDED_TO_TERMINAL : NOT_DELIVERED } @@ -98,8 +103,8 @@ function launchSubmitText(intent: AgentLaunchIntent): string | undefined { } /** - * The prompt a terminal's launch command should carry, which is an argv-mode agent's and only an - * argv-mode agent's. + * The prompt offered to a terminal's launch command, which is an argv-mode agent's and only an + * argv-mode agent's. An offer, not a decision: the surface reports whether the typed line took it. */ export function argvLaunchPrompt(intent: AgentLaunchIntent): string | undefined { const text = launchSubmitText(intent) diff --git a/src/main/agent-launch/agent-launch-surface-factories.ts b/src/main/agent-launch/agent-launch-surface-factories.ts index 867eac40ab6..e7d36f05ea1 100644 --- a/src/main/agent-launch/agent-launch-surface-factories.ts +++ b/src/main/agent-launch/agent-launch-surface-factories.ts @@ -25,8 +25,8 @@ export type AgentLaunchSurfaceFactory = { worktreeId: string agent: TuiAgent options?: Readonly> - /** Set only for an agent whose CLI takes the prompt on argv, so the text is in the process's - * arguments at exec time rather than raced into its composer afterwards. */ + /** Offered only for an agent whose CLI takes the prompt on argv. It rides the launch command + * only when the typed line can carry it; `promptRodeLaunchCommand` reports which happened. */ startupPrompt?: string /** Replaces the settings default for this launch only; `null` means no arguments at all. */ agentArgs?: string | null @@ -40,6 +40,8 @@ export type AgentLaunchSurfaceFactory = { /** The pane this create minted; a factory whose runtime reports none omits it, never invents. */ paneKey?: string warning?: string + /** Reported by the surface that built the typed line, never predicted by the executor. */ + promptRodeLaunchCommand?: boolean }> /** * Commits the launch text as the session's first turn, answering with the transcript row's id. @@ -56,12 +58,19 @@ export type AgentLaunchSurfaceFactory = { /** * Writes the launch text into a terminal agent's live PTY, answering whether it landed. * - * The other half of `startupPrompt`, for the two cases argv cannot serve: a `stdin-after-start` - * agent, whose CLI takes no prompt argument, and a reused terminal, whose process was already - * running before this launch existed. `false` for every failure, on the same rule the structured + * The other half of `startupPrompt`, for the cases the launch command cannot serve: a + * `stdin-after-start` agent, whose CLI takes no prompt argument; a prompt the typed line cannot + * carry; and a reused terminal, whose process was already running before this launch existed. `false` for every failure, on the same rule the structured * twin follows — a launch whose agent is running must not fail because its text did not land. */ - deliverTerminalPrompt?(args: { handle: string; prompt: AgentLaunchPrompt }): Promise + deliverTerminalPrompt?(args: { + handle: string + /** The launched agent, whose own readiness signal the write waits for. */ + agent: TuiAgent + /** False for a reused terminal, which has no fresh launch readiness to wait for. */ + freshLaunch: boolean + prompt: AgentLaunchPrompt + }): Promise } /** `fence` is the lease the create was admitted at, carried so the launch prompt's send can fill its @@ -95,8 +104,8 @@ export type AgentLaunchWorkspaceFactory = { * wait-for-setup gate for free. A structured launch has no startup command to sequence and * must await that gate explicitly instead. */ startupAgent: TuiAgent | undefined - /** Set only alongside a `startupAgent` whose CLI takes the prompt on argv: agent-first creation - * builds the startup command, so that is where an argv prompt belongs. */ + /** Offered only alongside a `startupAgent` whose CLI takes the prompt on argv: agent-first + * creation builds the startup command, so that is where the typed line is measured. */ startupPrompt?: string /** Inputs needed when this terminal is created as the worktree's startup surface. */ agentArgs?: string | null @@ -112,5 +121,7 @@ export type AgentLaunchWorkspaceFactory = { startupTerminalPaneKey?: string /** Created, but incomplete — surfaced on the launch result rather than dropped. */ warning?: string + /** Reported by the create that built the startup command's typed line. */ + promptRodeLaunchCommand?: boolean }> } diff --git a/src/main/agent-launch/host-agent-startup-attribution.test.ts b/src/main/agent-launch/host-agent-startup-attribution.test.ts index 612caab8001..d29c1cf7c0a 100644 --- a/src/main/agent-launch/host-agent-startup-attribution.test.ts +++ b/src/main/agent-launch/host-agent-startup-attribution.test.ts @@ -22,8 +22,9 @@ const LISTED: ReadonlyMap = new Map }) ) +// planStartupWithPromptCandidate wraps buildAgentStartupPlan in shared/, which this scan does not reach. const BUILDER_CALL = - /\b(?:buildAgentStartupPlan|buildAgentDraftLaunchPlan|buildAgentResumeStartupPlan)\s*\(/g + /\b(?:buildAgentStartupPlan|buildAgentDraftLaunchPlan|buildAgentResumeStartupPlan|planStartupWithPromptCandidate)\s*\(/g const DECIDE = 'Decide attribution: a fresh agent the host builds must carry ' + diff --git a/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts b/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts index 3d7f236917d..7c9df1e6bc2 100644 --- a/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts +++ b/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts @@ -7,7 +7,6 @@ import { describe, expect, it } from 'vitest' import { - AGENT_LAUNCH_RUNTIME_CAPABILITY, AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, AUTOMATION_CREATE_IDEMPOTENCY_RUNTIME_CAPABILITY, AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY, @@ -23,6 +22,7 @@ import { WORKTREE_VISIBILITY_SOURCE_DEFAULTS_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../shared/protocol-version' +import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability' import { ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES } from '../../shared/electron-remote-runtime-client-capabilities' import { remoteRuntimeClientCapabilities } from '../../shared/remote-runtime-client-capabilities' import { supportsAgentLaunch } from '../runtime/rpc/methods/agent-launch' diff --git a/src/main/ipc/desktop-renderer-runtime-capabilities.ts b/src/main/ipc/desktop-renderer-runtime-capabilities.ts index ba4b6b25472..f1eb84dca40 100644 --- a/src/main/ipc/desktop-renderer-runtime-capabilities.ts +++ b/src/main/ipc/desktop-renderer-runtime-capabilities.ts @@ -1,5 +1,4 @@ import { - AGENT_LAUNCH_RUNTIME_CAPABILITY, AGENT_SESSION_ACCEPTED_SEND_RUNTIME_CAPABILITY, AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, @@ -10,6 +9,7 @@ import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../shared/protocol-version' +import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability' import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability' /** diff --git a/src/main/ipc/runtime-agent-launch-capability.test.ts b/src/main/ipc/runtime-agent-launch-capability.test.ts index b88a3ca8b24..6c85455a324 100644 --- a/src/main/ipc/runtime-agent-launch-capability.test.ts +++ b/src/main/ipc/runtime-agent-launch-capability.test.ts @@ -6,10 +6,8 @@ */ import { beforeEach, describe, expect, it, vi } from 'vitest' -import { - AGENT_LAUNCH_RUNTIME_CAPABILITY, - type RuntimeCapability -} from '../../shared/protocol-version' +import type { RuntimeCapability } from '../../shared/protocol-version' +import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability' type AdvertisedClient = { clientKind?: 'mobile' | 'runtime' diff --git a/src/main/opencode/opencode-model-startup-plan.ts b/src/main/opencode/opencode-model-startup-plan.ts index f0878bd5177..84b0599ce88 100644 --- a/src/main/opencode/opencode-model-startup-plan.ts +++ b/src/main/opencode/opencode-model-startup-plan.ts @@ -1,5 +1,10 @@ import type { AgentStartupPlanInputs } from '../../shared/agent-startup-plan-inputs' -import { buildAgentDraftLaunchPlan, buildAgentStartupPlan } from '../../shared/tui-agent-startup' +import { + buildAgentDraftLaunchPlan, + buildAgentStartupPlan, + type AgentStartupPlan +} from '../../shared/tui-agent-startup' +import { planStartupWithPromptCandidate } from '../../shared/startup-line-prompt-carry' import { buildSleepingAgentLaunchConfig } from '../../shared/sleeping-agent-launch-config' import type { SleepingAgentLaunchConfig } from '../../shared/agent-session-resume' import type { TuiAgent } from '../../shared/tui-agent' @@ -162,6 +167,22 @@ export async function buildExecutionHostAgentStartupPlan( return plan } +/** `buildExecutionHostAgentStartupPlan` for a caller that delivers an uncarried prompt itself. */ +export async function planExecutionHostStartupWithPromptCandidate( + options: StartupScope & { + prompt: string + host: { shellName?: string; provesAgentInFront: boolean } + } +): Promise<{ plan: AgentStartupPlan | null; promptCarried: boolean }> { + const prepared = await prepareOpenCodeModelStartupInputs(options) + const offered = planStartupWithPromptCandidate(prepared.inputs, options.prompt, options.host) + if (offered.plan && prepared.launchConfig) { + offered.plan.launchConfig = prepared.launchConfig + offered.plan.sessionOptions = { ...options.inputs.sessionOptions } + } + return offered +} + export function assertOpenCodeModelLaunchPreferencesAbsent( agent: TuiAgent | undefined, preferences: Readonly> | undefined diff --git a/src/main/providers/startup-line-typed-length.live-shell.test.ts b/src/main/providers/startup-line-typed-length.live-shell.test.ts new file mode 100644 index 00000000000..e74816351b4 --- /dev/null +++ b/src/main/providers/startup-line-typed-length.live-shell.test.ts @@ -0,0 +1,152 @@ +/** + * Real-zsh proof for the typed launch lines `startup-line-prompt-carry.ts` lets a prompt ride: a + * multi-line line whose every line fits the per-line budget, up to the whole-line budget, reaches + * zsh intact through Orca's own wrapper, ready marker and startup write. + * + * Why a slow user config too: the write is released 1.5 s after spawn even without the marker, and + * then lands while the terminal is still line-buffered, where macOS keeps at most MAX_CANON bytes of + * one line. The per-line budget is what survives that; a single 1.1 KB line did not. + */ +import { spawnSync } from 'node:child_process' +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import * as pty from 'node-pty' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + createShellStartupOutputScanState, + scanShellStartupOutput +} from '../shell-startup-output-scanner' +import { selectShellStartupFeatures } from '../shell-startup-features' +import { isBracketedPasteSafeShell } from '../../shared/startup-command-submission' +import { + TYPED_STARTUP_LINE_PROMPT_BUDGET_BYTES, + ZSH_MULTI_LINE_STARTUP_LINE_BUDGET_BYTES +} from '../../shared/startup-line-prompt-carry' +import { + restoreUserDataPathAfterEach, + setTestUserDataPath +} from './local-pty-shell-ready-test-harness' + +function findZsh(): string { + if (process.platform === 'win32') { + return '' + } + return (spawnSync('sh', ['-c', 'command -v zsh'], { encoding: 'utf8' }).stdout ?? '').trim() +} + +const ZSH_PATH = findZsh() + +/** Lines of prompt text, each `perLine` bytes. */ +function promptLines(lineCount: number, perLine: number): string { + const words = 'Fix the failing checks on this branch, then explain what changed and why.' + return Array.from({ length: lineCount }, (_unused, index) => { + let line = `L${index}:` + while (line.length < perLine) { + line += ` ${words}` + } + return line.slice(0, perLine) + }).join('\n') +} + +let home = '' + +beforeEach(() => { + home = mkdtempSync(join(tmpdir(), 'orca-typed-line-home-')) + setTestUserDataPath(mkdtempSync(join(tmpdir(), 'orca-typed-line-ud-'))) +}) + +afterEach(() => { + rmSync(home, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }) +}) + +restoreUserDataPathAfterEach() + +/** Types `printf '%s' '' > out` the way a local pane types a launch line; returns what ran. */ +async function typeIntoZsh(text: string, userConfigSeconds: number): Promise { + if (userConfigSeconds > 0) { + writeFileSync(join(home, '.zshrc'), `sleep ${userConfigSeconds}\n`) + } + const out = join(home, 'out.txt') + const done = join(home, 'done.txt') + const { ensureShellReadyWrappers } = await import('./local-pty-shell-ready-wrapper-generation') + const { getShellLaunchConfig } = await import('./local-pty-shell-ready') + const { writeStartupCommandWhenShellReady, STARTUP_COMMAND_READY_MAX_WAIT_MS } = + await import('./local-pty-shell-ready-startup-command') + ensureShellReadyWrappers() + const env: Record = { PATH: '/usr/bin:/bin', HOME: home } + const launch = getShellLaunchConfig( + ZSH_PATH, + selectShellStartupFeatures({ + shellPath: ZSH_PATH, + env, + hasStartupCommand: true, + waitsForShellReady: true, + emitsStartupIdentity: false + }) + ) + expect(launch.supportsReadyMarker).toBe(true) + const proc = pty.spawn(ZSH_PATH, launch.args ?? [], { + name: 'xterm-256color', + cols: 200, + rows: 40, + cwd: home, + // Why ORCA_ORIG_ZDOTDIR: the user's config is read from the sandbox home, not the real one. + env: { ...env, ...launch.env, ORCA_ORIG_ZDOTDIR: home, TERM: 'xterm-256color' } + }) + const scan = createShellStartupOutputScanState() + let resolveReady: ((signal: { postMarkerBytesObserved: boolean }) => void) | null = null + const ready = new Promise<{ postMarkerBytesObserved: boolean }>((resolve) => { + resolveReady = resolve + }) + // The provider's cap (local-pty-shell-readiness-session.ts): a later marker releases the write. + const cap = setTimeout(() => { + resolveReady?.({ postMarkerBytesObserved: false }) + resolveReady = null + }, STARTUP_COMMAND_READY_MAX_WAIT_MS) + proc.onData((data) => { + if (resolveReady && scanShellStartupOutput(scan, data).ready) { + resolveReady({ postMarkerBytesObserved: true }) + resolveReady = null + } + }) + writeStartupCommandWhenShellReady( + ready, + proc, + `printf '%s' '${text}' > '${out}'; printf ok > '${done}'`, + () => {}, + { + bracketedPasteSafe: isBracketedPasteSafeShell({ + shellName: 'zsh', + waitsForShellReady: launch.supportsReadyMarker === true + }) + } + ) + const deadline = Date.now() + 20_000 + userConfigSeconds * 1000 + while (!existsSync(done) && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 50)) + } + clearTimeout(cap) + const ran = existsSync(out) ? readFileSync(out, 'utf8') : null + const exited = new Promise((resolve) => proc.onExit(() => resolve())) + proc.kill() + await Promise.race([exited, new Promise((resolve) => setTimeout(resolve, 2000))]) + return ran +} + +describe.skipIf(!ZSH_PATH)('a multi-line launch line typed into zsh', () => { + // Lines at the per-line budget, as many as the whole-line budget leaves room for. + const perLine = TYPED_STARTUP_LINE_PROMPT_BUDGET_BYTES - 1 + const text = promptLines( + Math.floor((ZSH_MULTI_LINE_STARTUP_LINE_BUDGET_BYTES - 200) / (perLine + 1)), + perLine + ) + + it('arrives whole through the ready barrier', async () => { + expect(await typeIntoZsh(text, 0)).toBe(text) + }, 60_000) + + it('arrives whole when the user config outlasts the barrier and the write lands early', async () => { + expect(await typeIntoZsh(text, 3)).toBe(text) + }, 60_000) +}) diff --git a/src/main/runtime/__fixtures__/readiness-census/synthetic--gemini.json b/src/main/runtime/__fixtures__/readiness-census/synthetic--gemini.json index dc869633ad6..322410731f3 100644 --- a/src/main/runtime/__fixtures__/readiness-census/synthetic--gemini.json +++ b/src/main/runtime/__fixtures__/readiness-census/synthetic--gemini.json @@ -29,20 +29,20 @@ "title=working-spinner status=blocked-fresh screen=present fg=agent clock=clockless": "verdict=pending:closed wait=pending", "title=working-spinner status=blocked-stale screen=present fg=agent clock=clocked": "now=working edge=working quiet=working wait=pending", "title=working-spinner status=blocked-stale screen=present fg=agent clock=clockless": "verdict=working wait=pending", - "title=name-only status=none screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=none screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=done-fresh screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=done-fresh screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=done-stale screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=done-stale screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=working-fresh screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=working-fresh screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=working-stale screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=working-stale screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=blocked-fresh screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=blocked-fresh screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=blocked-stale screen=present fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=blocked-stale screen=present fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", + "title=name-only status=none screen=present fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=none screen=present fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=done-fresh screen=present fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=done-fresh screen=present fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=done-stale screen=present fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=done-stale screen=present fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=working-fresh screen=present fg=agent clock=clocked": "now=working edge=working quiet=working wait=pending", + "title=name-only status=working-fresh screen=present fg=agent clock=clockless": "verdict=working wait=pending", + "title=name-only status=working-stale screen=present fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=working-stale screen=present fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=blocked-fresh screen=present fg=agent clock=clocked": "now=pending:closed edge=pending:closed quiet=pending:closed wait=pending", + "title=name-only status=blocked-fresh screen=present fg=agent clock=clockless": "verdict=pending:closed wait=pending", + "title=name-only status=blocked-stale screen=present fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=blocked-stale screen=present fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", "title=none status=none screen=present fg=agent clock=clocked": "now=pending:closed edge=pending:closed quiet=pending:closed wait=pending", "title=none status=none screen=present fg=agent clock=clockless": "verdict=pending:closed wait=pending", "title=none status=done-fresh screen=present fg=agent clock=clocked": "now=pending:closed edge=pending:closed quiet=pending:closed wait=pending", @@ -57,16 +57,16 @@ "title=none status=blocked-fresh screen=present fg=agent clock=clockless": "verdict=pending:closed wait=pending", "title=none status=blocked-stale screen=present fg=agent clock=clocked": "now=pending:closed edge=pending:closed quiet=pending:closed wait=pending", "title=none status=blocked-stale screen=present fg=agent clock=clockless": "verdict=pending:closed wait=pending", - "title=name-only status=none screen=present fg=shell clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=none screen=present fg=shell clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=none screen=untrusted fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=none screen=untrusted fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=none screen=untrusted fg=shell clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=none screen=untrusted fg=shell clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=none screen=absent fg=agent clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=none screen=absent fg=agent clock=clockless": "verdict=ready-strong wait=ready@start", - "title=name-only status=none screen=absent fg=shell clock=clocked": "now=ready-strong edge=ready-strong quiet=ready-strong wait=ready@start", - "title=name-only status=none screen=absent fg=shell clock=clockless": "verdict=ready-strong wait=ready@start", + "title=name-only status=none screen=present fg=shell clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=none screen=present fg=shell clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=none screen=untrusted fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=none screen=untrusted fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=none screen=untrusted fg=shell clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=none screen=untrusted fg=shell clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=none screen=absent fg=agent clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=none screen=absent fg=agent clock=clockless": "verdict=ready-weak wait=ready@start", + "title=name-only status=none screen=absent fg=shell clock=clocked": "now=ready-weak edge=ready-weak quiet=ready-weak wait=ready@start", + "title=name-only status=none screen=absent fg=shell clock=clockless": "verdict=ready-weak wait=ready@start", "title=none status=none screen=present fg=shell clock=clocked": "now=pending:closed edge=pending:closed quiet=pending:closed wait=pending", "title=none status=none screen=present fg=shell clock=clockless": "verdict=pending:closed wait=pending", "title=none status=none screen=untrusted fg=agent clock=clocked": "now=pending:closed edge=pending:closed quiet=pending:closed wait=pending", diff --git a/src/main/runtime/__fixtures__/readiness-census/transcript--zsh-prompt-runs-command@unknown.json b/src/main/runtime/__fixtures__/readiness-census/transcript--zsh-prompt-runs-command@unknown.json new file mode 100644 index 00000000000..6b069e10350 --- /dev/null +++ b/src/main/runtime/__fixtures__/readiness-census/transcript--zsh-prompt-runs-command@unknown.json @@ -0,0 +1,7 @@ +{ + "description": "non-agent recording at 80x24 replayed on the unknown pane, one entry per chunk", + "observations": { + "clocked": ["0-4: now=pending:open edge=pending:open quiet=pending:open wait=pending"], + "clockless": ["0-4: verdict=pending:open wait=pending"] + } +} diff --git a/src/main/runtime/__fixtures__/zsh-prompt-runs-command.meta.json b/src/main/runtime/__fixtures__/zsh-prompt-runs-command.meta.json new file mode 100644 index 00000000000..1e7291a03a8 --- /dev/null +++ b/src/main/runtime/__fixtures__/zsh-prompt-runs-command.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-30T12:00:00.000Z", + "platform": "darwin", + "command": ["/bin/zsh", "-f", "-i"], + "cols": 80, + "rows": 24, + "note": "zsh 5.9, macOS arm64, clean environment, PS1='%# '. The prompt turns bracketed paste on (ESC[?2004h), `sleep 1` is typed and run, which turns it off (ESC[?2004l), and the next prompt turns it on again: the bytes a launch sees from the shell before its agent starts, and after an agent exits", + "exitCode": null +} diff --git a/src/main/runtime/__fixtures__/zsh-prompt-runs-command.txt b/src/main/runtime/__fixtures__/zsh-prompt-runs-command.txt new file mode 100644 index 00000000000..eacea47373f --- /dev/null +++ b/src/main/runtime/__fixtures__/zsh-prompt-runs-command.txt @@ -0,0 +1,2 @@ +% % [?2004hssleep 1[?2004l +% % [?2004h \ No newline at end of file diff --git a/src/main/runtime/agent-launch-typed-line-shell.test.ts b/src/main/runtime/agent-launch-typed-line-shell.test.ts new file mode 100644 index 00000000000..9d17d07a49b --- /dev/null +++ b/src/main/runtime/agent-launch-typed-line-shell.test.ts @@ -0,0 +1,47 @@ +import { describe, expect, it } from 'vitest' +import { + launchHostProvesAgentInFront, + nameLocalTypedLineShell +} from './agent-launch-typed-line-shell' + +describe('naming the shell a local launch line is typed into', () => { + it('takes the request’s shell first, then the setting, then SHELL, as the spawn does', () => { + const base = { isRemote: false, platform: 'darwin' as const, envShell: '/bin/bash' } + expect( + nameLocalTypedLineShell({ + ...base, + shellOverride: '/opt/homebrew/bin/fish', + defaultShellSetting: '/bin/zsh' + }) + ).toBe('fish') + expect(nameLocalTypedLineShell({ ...base, defaultShellSetting: ' /bin/zsh ' })).toBe('zsh') + expect(nameLocalTypedLineShell(base)).toBe('bash') + expect(nameLocalTypedLineShell({ ...base, envShell: '' })).toBe('zsh') + }) + + it('names none for a remote host, whose relay picks its own login shell', () => { + expect( + nameLocalTypedLineShell({ isRemote: true, platform: 'darwin', envShell: '/bin/zsh' }) + ).toBeUndefined() + }) + + it('names none on Windows, where the pane may be cmd, PowerShell, Git Bash or WSL', () => { + expect( + nameLocalTypedLineShell({ isRemote: false, platform: 'win32', envShell: '/bin/zsh' }) + ).toBeUndefined() + }) +}) + +describe('whether a launch host can prove its agent is in front before a paste', () => { + it.each([ + ['a local macOS host', false, 'darwin', 'darwin', true], + ['a local Linux host', false, 'linux', 'linux', true], + ['a local Windows host', false, 'win32', 'win32', false], + // The pane runs in the distro, but the reads run on the Windows host. + ['a local WSL pane', false, 'linux', 'win32', false], + ['an SSH Linux host from Windows', true, 'linux', 'win32', true], + ['an SSH Windows host', true, 'win32', 'darwin', false] + ] as const)('%s: %s', (_label, isRemote, launchPlatform, hostPlatform, proves) => { + expect(launchHostProvesAgentInFront({ isRemote, launchPlatform, hostPlatform })).toBe(proves) + }) +}) diff --git a/src/main/runtime/agent-launch-typed-line-shell.ts b/src/main/runtime/agent-launch-typed-line-shell.ts new file mode 100644 index 00000000000..f733ade3385 --- /dev/null +++ b/src/main/runtime/agent-launch-typed-line-shell.ts @@ -0,0 +1,42 @@ +import { basename } from 'node:path' + +/** + * The shell a local pane's startup line gets typed into, named before the spawn the way the spawn + * picks it: the request's shell, the default-shell setting, then `SHELL` + * (`ipc/pty/runtime/spawn-preflight.ts`, then the local or daemon launch plan). + * + * Undefined where the host cannot name it: a remote host, whose relay picks its own login shell, + * and Windows, whose pane may be cmd, PowerShell, Git Bash or a WSL distro. + */ +export function nameLocalTypedLineShell(args: { + isRemote: boolean + shellOverride?: string + defaultShellSetting?: string + platform?: NodeJS.Platform + envShell?: string +}): string | undefined { + if (args.isRemote || (args.platform ?? process.platform) === 'win32') { + return undefined + } + const shellPath = + args.shellOverride?.trim() || + args.defaultShellSetting?.trim() || + (args.envShell ?? process.env.SHELL) || + '/bin/zsh' + return basename(shellPath).toLowerCase() +} + +/** + * Whether the execution host can prove a launched agent holds its terminal before a paste + * (`launched-agent-foreground`). A Windows host cannot, and a local WSL pane runs on one; an SSH + * host is judged by its own platform. + */ +export function launchHostProvesAgentInFront(args: { + isRemote: boolean + launchPlatform: NodeJS.Platform + hostPlatform?: NodeJS.Platform +}): boolean { + return args.isRemote + ? args.launchPlatform !== 'win32' + : (args.hostPlatform ?? process.platform) !== 'win32' +} diff --git a/src/main/runtime/agent-prompt-submission-runtime-test-fixture.ts b/src/main/runtime/agent-prompt-submission-runtime-test-fixture.ts index 69470ac6148..7db463bf2fa 100644 --- a/src/main/runtime/agent-prompt-submission-runtime-test-fixture.ts +++ b/src/main/runtime/agent-prompt-submission-runtime-test-fixture.ts @@ -1,3 +1,4 @@ +import { vi } from 'vitest' import type { TuiAgent } from '../../shared/tui-agent' import { OrcaRuntimeService } from './orca-runtime' import { makeStore } from './runtime-rpc-worktree-store-fixtures' @@ -24,5 +25,7 @@ export async function createAgentPromptSubmissionRuntime( const terminal = await runtime.createTerminal(`path:${AGENT_PROMPT_TEST_WORKTREE_PATH}`, { launchAgent }) + // The pane holds a live agent, which its host would find in front of its terminal. + vi.spyOn(runtime, 'readLaunchedAgentForeground').mockResolvedValue('agent') return { runtime, handle: terminal.handle, writes } } diff --git a/src/main/runtime/agent-session-operation-admission.ts b/src/main/runtime/agent-session-operation-admission.ts index f16d48efb9d..c83c6d9f006 100644 --- a/src/main/runtime/agent-session-operation-admission.ts +++ b/src/main/runtime/agent-session-operation-admission.ts @@ -193,6 +193,23 @@ export function claimAgentSessionOperationInto( return claimed.claim } +/** Whether an admission left the right to run open, so the same transaction should claim it. */ +export type ClaimAfterAdmission = (decision: AgentSessionOperationDecision) => boolean + +/** Admission and, when `claimAfter` says so, the claim, in one transaction: the same swap as + * `claimAgentSessionOperationInto`, with one durable write instead of two. */ +export function admitAndClaimAgentSessionOperationInto( + state: { operations: Map }, + args: AgentSessionOperationAdmission, + claimAfter: ClaimAfterAdmission +): { decision: AgentSessionOperationDecision; claim: AgentSessionOperationClaim | null } { + const decision = admitAgentSessionOperationInto(state, args) + return { + decision, + claim: claimAfter(decision) ? claimAgentSessionOperationInto(state, args) : null + } +} + export function settleAgentSessionOperationInto( state: { operations: Map }, args: { callerKey?: string; operationId: string; outcome: AgentSessionOperationOutcome } diff --git a/src/main/runtime/agent-session-record-store.ts b/src/main/runtime/agent-session-record-store.ts index 3ef04041a43..e2e2da7e1fe 100644 --- a/src/main/runtime/agent-session-record-store.ts +++ b/src/main/runtime/agent-session-record-store.ts @@ -21,6 +21,8 @@ import { evaluateAgentSessionMutationOperation, admitAgentSessionOperationInto, claimAgentSessionOperationInto, + admitAndClaimAgentSessionOperationInto, + type ClaimAfterAdmission, settleAgentSessionOperationInto, type AgentSessionMutationOperationAdmission, type AgentSessionOperationAdmission @@ -298,6 +300,12 @@ export class AgentSessionRecordStore { }): Promise => this.transact((draft) => claimAgentSessionOperationInto(draft, args)) + /** Admission and, when `claimAfter` allows, the claim: one durable write before the effect. */ + admitAndClaimOperation = ( + args: AgentSessionOperationAdmission, + claimAfter: ClaimAfterAdmission + ) => this.transact((draft) => admitAndClaimAgentSessionOperationInto(draft, args, claimAfter)) + async recordOperationOutcome(args: AgentSessionOperationSettlement): Promise { await this.transact((draft) => settleAgentSessionOperationInto(draft, args)) } diff --git a/src/main/runtime/agent-terminal-startup-prompt.test.ts b/src/main/runtime/agent-terminal-startup-prompt.test.ts index 8a9f9648147..cdf6cd16dbb 100644 --- a/src/main/runtime/agent-terminal-startup-prompt.test.ts +++ b/src/main/runtime/agent-terminal-startup-prompt.test.ts @@ -17,7 +17,10 @@ vi.mock('electron', () => ({ app: { getPath: vi.fn(() => '/tmp') } })) -function runtimeWithAgentLaunch(): { +// The shell the host names for a local line; bash takes no multi-line line, zsh does. +function runtimeWithAgentLaunch( + options: { terminalDefaultShell?: string; connectionId?: string } = {} +): { runtime: OrcaRuntimeService spawn: ReturnType } { @@ -27,11 +30,13 @@ function runtimeWithAgentLaunch(): { store: { getSettings: () => Record } resolveTerminalWorkspaceLaunchScope: (selector: string) => Promise } - internal.store = { getSettings: () => ({}) } + internal.store = { + getSettings: () => ({ terminalDefaultShell: options.terminalDefaultShell ?? '/bin/bash' }) + } vi.spyOn(internal, 'resolveTerminalWorkspaceLaunchScope').mockResolvedValue({ id: 'wt-1', path: '/repo/app', - connectionId: null, + connectionId: options.connectionId ?? null, repo: null, folderWorkspace: null }) @@ -52,16 +57,80 @@ function spawnedCommand(spawn: ReturnType): string { } describe('a terminal create that is handed a launch prompt', () => { - it('folds an argv agent’s prompt into the command it spawns', async () => { + it('folds an argv agent’s prompt into the command it spawns, and says it did', async () => { const { runtime, spawn } = runtimeWithAgentLaunch() + const onStartupPromptCarry = vi.fn() await runtime.createTerminal('id:wt-1', { startupAgent: 'claude', - startupPrompt: 'summarize the diff' + startupPrompt: 'summarize the diff', + onStartupPromptCarry }) expect(spawnedCommand(spawn)).toContain('summarize the diff') expect(spawn).toHaveBeenCalledWith(expect.objectContaining({ launchAgent: 'claude' })) + expect(onStartupPromptCarry).toHaveBeenCalledWith(true) + }) + + it('starts clean, and says so, when the typed line cannot carry a multi-line prompt', async () => { + const { runtime, spawn } = runtimeWithAgentLaunch() + const onStartupPromptCarry = vi.fn() + + await runtime.createTerminal('id:wt-1', { + startupAgent: 'claude', + startupPrompt: 'summarize the diff\nthen list the risks', + onStartupPromptCarry + }) + + // Typed into a shell, each newline would be Enter; the caller pastes it once the agent is up. + expect(spawnedCommand(spawn)).toContain('claude') + expect(spawnedCommand(spawn)).not.toContain('summarize') + expect(onStartupPromptCarry).toHaveBeenCalledWith(false) + }) + + it('carries a short-lined multi-line prompt on a local zsh line, so the agent starts with it', async () => { + const { runtime, spawn } = runtimeWithAgentLaunch({ terminalDefaultShell: '/bin/zsh' }) + const onStartupPromptCarry = vi.fn() + + await runtime.createTerminal('id:wt-1', { + startupAgent: 'claude', + startupPrompt: 'summarize the diff\nthen list the risks', + onStartupPromptCarry + }) + + expect(spawnedCommand(spawn)).toContain('summarize the diff\nthen list the risks') + expect(onStartupPromptCarry).toHaveBeenCalledWith(true) + }) + + it('starts clean on a remote host, whose shell this host cannot name', async () => { + const { runtime, spawn } = runtimeWithAgentLaunch({ + terminalDefaultShell: '/bin/zsh', + connectionId: 'ssh-1' + }) + const onStartupPromptCarry = vi.fn() + + await runtime.createTerminal('id:wt-1', { + startupAgent: 'claude', + startupPrompt: 'summarize the diff\nthen list the risks', + onStartupPromptCarry + }) + + expect(spawnedCommand(spawn)).not.toContain('summarize') + expect(onStartupPromptCarry).toHaveBeenCalledWith(false) + }) + + it('starts Hermes clean instead of refusing when its env budget cannot hold the prompt', async () => { + const { runtime, spawn } = runtimeWithAgentLaunch() + const onStartupPromptCarry = vi.fn() + + await runtime.createTerminal('id:wt-1', { + startupAgent: 'hermes', + startupPrompt: 'x'.repeat(30_000), + onStartupPromptCarry + }) + + expect(spawn).toHaveBeenCalledTimes(1) + expect(onStartupPromptCarry).toHaveBeenCalledWith(false) }) it('still builds a bare agent launch when no prompt is handed to it', async () => { diff --git a/src/main/runtime/agent-transcript-pane-test-harness.ts b/src/main/runtime/agent-transcript-pane-test-harness.ts index 8fa7701696a..89899702c7c 100644 --- a/src/main/runtime/agent-transcript-pane-test-harness.ts +++ b/src/main/runtime/agent-transcript-pane-test-harness.ts @@ -2,6 +2,7 @@ import { vi } from 'vitest' import { OrcaRuntimeService } from './orca-runtime' import type { TuiAgent } from '../../shared/tui-agent' +import type { PtyProcessInspection } from '../providers/pty-process-inspection' const TRANSCRIPT_PANE_LEAF_ID = '11111111-1111-4111-8111-111111111111' const TRANSCRIPT_PANE_TAB_ID = 'tab-1' @@ -15,11 +16,24 @@ export type TranscriptPaneOptions = { launchAgent?: TuiAgent /** Set for a pane whose PTY lives on an SSH host or WSL distro rather than locally. */ connectionId?: string + /** The remote host of a `connectionId` pane is Windows. */ + remoteWindowsHost?: boolean /** Simulates a PTY controller whose foreground probe never settles. */ foregroundProbeHangs?: boolean onForegroundProbe?: () => void /** PTY grid the controller reports; the runtime's emulator otherwise defaults to 80x24. */ size?: { cols: number; rows: number } + /** What a fresh foreground scan finds, where it differs from the cached foreground read. */ + confirmedForegroundProcess?: string | null + onForegroundScan?: () => void + /** What the host's process inspection answers; absent for a host without one. */ + processInspection?: PtyProcessInspection + onProcessInspection?: () => void + /** What the host's shell-foreground check answers: the spawned shell holds the foreground. */ + shellForegroundProven?: boolean + /** The pane's root process the provider reports; absent for a provider without an inventory. */ + paneRootPid?: number + onShellForegroundProof?: () => void } export async function createTranscriptPane( @@ -27,16 +41,21 @@ export async function createTranscriptPane( runtimeDeps?: ConstructorParameters[2] ): Promise<{ runtime: OrcaRuntimeService; handle: string }> { const runtime = new OrcaRuntimeService(null, undefined, runtimeDeps) + // The runtime reads a remote pane's host OS from its worktree path. + const worktreeId = options.remoteWindowsHost + ? 'repo-1::C:\\repo\\app' + : TRANSCRIPT_PANE_WORKTREE_ID const internals = runtime as unknown as { resolveTerminalWorkspaceLaunchScope: (selector: string) => Promise } vi.spyOn(internals, 'resolveTerminalWorkspaceLaunchScope').mockResolvedValue({ - id: TRANSCRIPT_PANE_WORKTREE_ID, + id: worktreeId, path: '/repo/app', connectionId: options.connectionId ?? null, repo: null, folderWorkspace: null }) + const processInspection = options.processInspection runtime.setPtyController({ spawn: vi.fn().mockResolvedValue({ id: TRANSCRIPT_PANE_PTY_ID, incarnationId: 'inc-1' }), write: () => true, @@ -47,9 +66,45 @@ export async function createTranscriptPane( return options.foregroundProbeHangs === true ? new Promise(() => {}) : Promise.resolve(options.foregroundProcess) - } + }, + ...(options.confirmedForegroundProcess !== undefined + ? { + confirmForegroundProcess: async () => { + options.onForegroundScan?.() + return options.confirmedForegroundProcess ?? null + } + } + : {}), + ...(processInspection + ? { + inspectProcess: async () => { + options.onProcessInspection?.() + return processInspection + } + } + : {}), + ...(options.paneRootPid !== undefined + ? { + listProcesses: async () => [ + { + id: TRANSCRIPT_PANE_PTY_ID, + rootProcessId: options.paneRootPid, + cwd: '/repo/app', + title: 'Terminal' + } + ] + } + : {}), + ...(options.shellForegroundProven !== undefined + ? { + confirmShellForeground: async () => { + options.onShellForegroundProof?.() + return options.shellForegroundProven === true + } + } + : {}) }) - const terminal = await runtime.createTerminal(`id:${TRANSCRIPT_PANE_WORKTREE_ID}`, { + const terminal = await runtime.createTerminal(`id:${worktreeId}`, { tabId: TRANSCRIPT_PANE_TAB_ID, leafId: TRANSCRIPT_PANE_LEAF_ID, title: 'Terminal' @@ -59,7 +114,7 @@ export async function createTranscriptPane( tabs: [ { tabId: TRANSCRIPT_PANE_TAB_ID, - worktreeId: TRANSCRIPT_PANE_WORKTREE_ID, + worktreeId, title: 'Terminal', activeLeafId: TRANSCRIPT_PANE_LEAF_ID, layout: null @@ -68,7 +123,7 @@ export async function createTranscriptPane( leaves: [ { tabId: TRANSCRIPT_PANE_TAB_ID, - worktreeId: TRANSCRIPT_PANE_WORKTREE_ID, + worktreeId, leafId: TRANSCRIPT_PANE_LEAF_ID, paneRuntimeId: 1, ptyId: TRANSCRIPT_PANE_PTY_ID, @@ -77,7 +132,7 @@ export async function createTranscriptPane( ] }) if (options.launchAgent) { - runtime.registerPty(TRANSCRIPT_PANE_PTY_ID, TRANSCRIPT_PANE_WORKTREE_ID, null, { + runtime.registerPty(TRANSCRIPT_PANE_PTY_ID, worktreeId, options.connectionId ?? null, { tabId: TRANSCRIPT_PANE_TAB_ID, leafId: TRANSCRIPT_PANE_LEAF_ID, incarnationId: 'inc-1', diff --git a/src/main/runtime/launched-agent-composer-readiness-claude.test.ts b/src/main/runtime/launched-agent-composer-readiness-claude.test.ts new file mode 100644 index 00000000000..60557680044 --- /dev/null +++ b/src/main/runtime/launched-agent-composer-readiness-claude.test.ts @@ -0,0 +1,386 @@ +/** + * A freshly launched Claude's first input, replayed from captured transcripts + * (`__fixtures__/claude-dialog-trust-workspace*.txt`). + * + * The launch pastes on the signal the desktop's own paste used: bracketed paste turned on, then a + * quiet render. Claude's first-launch trust dialog renders in that same mode, so the quiet window + * also settles over it, and only the screen check keeps the prompt out of the dialog. + */ + +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { createTranscriptPane, TRANSCRIPT_PANE_PTY_ID } from './agent-transcript-pane-test-harness' +import { + waitForLaunchedAgentComposer, + waitForWorkerStartComposer +} from './launched-agent-composer-readiness' +import { resolveRemoteForegroundEvidence } from '../providers/agent-foreground-process' +import type { ProcessTableRow } from '../../shared/process-table-snapshot' +import type * as TerminalForegroundGroup from './terminal-foreground-group' + +// What `ps` limited to the pane's terminal answers; the verdict over it stays the real one. +const paneTerminal = vi.hoisted(() => { + const state: { rows: ProcessTableRow[] | null } = { rows: null } + return state +}) +vi.mock('./terminal-foreground-group', async (importOriginal) => ({ + ...(await importOriginal()), + readTerminalProcessRows: vi.fn(async () => paneTerminal.rows) +})) + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +/** The desktop paste's quiet window after bracketed paste, which the launch now shares. */ +const QUIET_WINDOW_MS = 1_500 + +function readCapture(name: string): { data: string; size: { cols: number; rows: number } } { + const base = join(__dirname, '__fixtures__', name) + const meta: { cols: number; rows: number } = JSON.parse(readFileSync(`${base}.meta.json`, 'utf8')) + return { data: readFileSync(`${base}.txt`, 'utf8'), size: { cols: meta.cols, rows: meta.rows } } +} + +/** Starts the launch wait on an empty pane, then streams the capture in, as a live launch does. */ +async function launchAndStream( + name: string, + timeoutMs: number, + pane: Partial[0]> = {} +) { + const { data, size } = readCapture(name) + const { runtime, handle } = await createTranscriptPane({ + paneTitle: 'Claude Code', + foregroundProcess: 'claude', + launchAgent: 'claude', + size, + data: '', + ...pane + }) + // Pane creation awaits real timers; the wait and its quiet window use the virtual clock. + vi.useFakeTimers() + const ready = waitForLaunchedAgentComposer(runtime, handle, 'claude', timeoutMs) + const settled = vi.fn() + ready.then(settled, settled) + runtime.onPtyData(TRANSCRIPT_PANE_PTY_ID, data, Date.now()) + return { ready, settled } +} + +describe('launch readiness for a freshly launched Claude', () => { + afterEach(() => { + vi.useRealTimers() + }) + + it('claude-dialog-trust-workspace-answered: reads the composer after the answered dialog as ready', async () => { + const { data } = readCapture('claude-dialog-trust-workspace-answered') + // Presence precondition: the capture turns bracketed paste on and ends on the idle composer. + expect(data).toContain('\x1b[?2004h') + const { ready, settled } = await launchAndStream( + 'claude-dialog-trust-workspace-answered', + 60_000 + ) + + await vi.advanceTimersByTimeAsync(QUIET_WINDOW_MS + 100) + expect(settled).toHaveBeenCalled() + await expect(ready).resolves.toMatchObject({ satisfied: true }) + }) + + it('claude-dialog-trust-workspace-answered: over SSH, settles on the quiet window, not the 8 s fallback', async () => { + const { ready, settled } = await launchAndStream( + 'claude-dialog-trust-workspace-answered', + 60_000, + // The relay offers no scan or shell check; either claiming a shell would refuse this signal. + { connectionId: 'ssh-1', confirmedForegroundProcess: 'zsh', shellForegroundProven: true } + ) + + await vi.advanceTimersByTimeAsync(QUIET_WINDOW_MS + 300) + expect(settled).toHaveBeenCalled() + await expect(ready).resolves.toMatchObject({ satisfied: true }) + }) + + it('claude-dialog-trust-workspace: never reads the trust dialog as the composer', async () => { + const { ready, settled } = await launchAndStream('claude-dialog-trust-workspace', 60_000) + + // Reported inside the desktop paste's budget: the quiet window settles over the dialog, the + // screen check refuses it, and the idle wait names it. + await vi.advanceTimersByTimeAsync(4_000) + expect(settled).toHaveBeenCalled() + await expect(ready).resolves.toMatchObject({ + satisfied: false, + blockedReason: 'agent-trust-workspace' + }) + }) +}) + +describe('a fresh orchestration worker start for Claude', () => { + afterEach(() => { + vi.useRealTimers() + }) + + // Main's worker start pasted on this cue; the launch's quiet window put the brief 1.6 s later. + it('settles on Claude’s ready title, before the launch paste’s quiet window', async () => { + const { data, size } = readCapture('claude-dialog-trust-workspace-answered') + const { runtime, handle } = await createTranscriptPane({ + paneTitle: 'Claude Code', + foregroundProcess: 'claude', + launchAgent: 'claude', + size, + data: '' + }) + vi.useFakeTimers() + const ready = waitForWorkerStartComposer(runtime, handle, 'claude', 60_000) + const settled = vi.fn() + ready.then(settled, settled) + runtime.onPtyData(TRANSCRIPT_PANE_PTY_ID, data, Date.now()) + + await vi.advanceTimersByTimeAsync(QUIET_WINDOW_MS / 2) + expect(settled).toHaveBeenCalled() + await expect(ready).resolves.toMatchObject({ satisfied: true }) + }) +}) + +describe('what holds a launched agent’s terminal', () => { + const originalPlatform = Object.getOwnPropertyDescriptor(process, 'platform')! + const setPlatform = (value: NodeJS.Platform): void => { + Object.defineProperty(process, 'platform', { configurable: true, value }) + } + afterEach(() => { + Object.defineProperty(process, 'platform', originalPlatform) + }) + + describe('macOS and Linux: the pane terminal’s own foreground group decides', () => { + beforeEach(() => { + setPlatform('darwin') + paneTerminal.rows = null + }) + + // Captured with `ps -o pid=,ppid=,pgid=,tpgid=,stat=,command=` on a pane spawned the way a macOS + // pane is, under `login`: zsh holds the terminal in its own group, never the root's. + const login = (tpgid: number): ProcessTableRow => ({ + pid: 60404, + ppid: 60394, + pgid: 60404, + tpgid, + stat: 'Ss', + command: '/usr/bin/login -flpq user /bin/bash --noprofile --norc -p -c' + }) + const zshAt = (tpgid: number): ProcessTableRow => ({ + pid: 60406, + ppid: 60404, + pgid: 60406, + tpgid, + stat: tpgid === 60406 ? 'S+' : 'S', + command: '-/bin/zsh -f' + }) + + it.each([ + // A crashed stub's name can outlive it in the cached read; the rows show zsh back in front. + ['zsh back at its prompt', [login(60406), zshAt(60406)], 'shell'], + [ + 'claude in front', + [ + login(60500), + zshAt(60500), + { pid: 60500, ppid: 60406, pgid: 60500, tpgid: 60500, stat: 'S+', command: 'claude' } + ], + 'agent' + ] + ] as const)( + '%s: answers from the rows, never the cached name or a whole-machine scan', + async (_label, rows, found) => { + paneTerminal.rows = [...rows] + const scan = vi.fn() + const inspection = vi.fn() + const { runtime } = await createTranscriptPane({ + paneTitle: 'Claude Code', + foregroundProcess: 'python3', + confirmedForegroundProcess: 'claude', + onForegroundScan: scan, + processInspection: { foregroundProcess: 'claude', hasChildProcesses: true }, + onProcessInspection: inspection, + paneRootPid: 60404, + launchAgent: 'claude', + data: '' + }) + + await expect( + runtime.readLaunchedAgentForeground(TRANSCRIPT_PANE_PTY_ID, 'claude') + ).resolves.toBe(found) + expect(scan).not.toHaveBeenCalled() + expect(inspection).not.toHaveBeenCalled() + } + ) + + it.each([ + ['without the pane’s root process', undefined, [login(60406), zshAt(60406)]], + ['when ps cannot read the pane’s terminal', 60404, null] + ] as const)('proves nothing %s', async (_label, paneRootPid, rows) => { + paneTerminal.rows = rows ? [...rows] : null + const { runtime } = await createTranscriptPane({ + paneTitle: 'Claude Code', + foregroundProcess: 'claude', + ...(paneRootPid ? { paneRootPid } : {}), + launchAgent: 'claude', + data: '' + }) + + await expect( + runtime.readLaunchedAgentForeground(TRANSCRIPT_PANE_PTY_ID, 'claude') + ).resolves.toBe('unknown') + }) + }) + + // A tcsh or nu launch line runs the agent from `/bin/sh '