diff --git a/src/main/agent-launch/agent-launch-executor.test.ts b/src/main/agent-launch/agent-launch-executor.test.ts index 99c8f3d40e1..8829069a451 100644 --- a/src/main/agent-launch/agent-launch-executor.test.ts +++ b/src/main/agent-launch/agent-launch-executor.test.ts @@ -615,7 +615,7 @@ describe('the surface is published as the launch stands, before its prompt is de return { launch, published } } - it('records a prompt still owed as not delivered, then delivers it', async () => { + it('records a prompt still owed as unconfirmed, then delivers it', async () => { const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: false }) const result = await launch.run(PROMPTED_EXISTING) @@ -626,12 +626,25 @@ describe('the surface is published as the launch stands, before its prompt is de outcome: { kind: 'terminal', handle: 'term_1' }, worktreeId: 'wt-7', receipt: result.receipt, - prompt: { delivery: 'submit', outcome: 'not-delivered' } + // A host that stops mid-paste cannot say whether it landed, so it must not say "not sent". + prompt: { delivery: 'submit', outcome: 'unconfirmed' } } ]) expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) }) + it('records a draft as not delivered, since the host never delivers one', async () => { + const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: false }) + + const result = await launch.run({ + ...PROMPTED_EXISTING, + prompt: { text: 'fix the build', delivery: 'draft' } + }) + + expect(published[0]?.prompt).toEqual({ delivery: 'draft', outcome: 'not-delivered' }) + expect(result.prompt).toEqual({ delivery: 'draft', outcome: 'not-delivered' }) + }) + it('records a prompt the launch command carried as already handed over', async () => { const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: true }) @@ -655,7 +668,7 @@ describe('the surface is published as the launch stands, before its prompt is de ]) expect(published[0]).toEqual({ ...result, - prompt: { delivery: 'submit', outcome: 'not-delivered' } + prompt: { delivery: 'submit', outcome: 'unconfirmed' } }) expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'journaled', messageId: 'msg-1' }) }) diff --git a/src/main/agent-launch/agent-launch-executor.ts b/src/main/agent-launch/agent-launch-executor.ts index a01abf8898f..95bb7b0ec16 100644 --- a/src/main/agent-launch/agent-launch-executor.ts +++ b/src/main/agent-launch/agent-launch-executor.ts @@ -77,8 +77,9 @@ export type AgentLaunchExecution = { /** * The launch as it stands once its surface exists: a complete result whose prompt receipt says only - * what creation itself settled — carried on the launch command, or not (yet) delivered. Complete so - * a host that dies during the delivery still leaves a truthful answer behind. + * what creation itself settled — carried on the launch command, a draft the host never delivers, or + * a submit still `unconfirmed`. Complete so a host that dies during the delivery still leaves a + * truthful answer behind. */ export type AgentLaunchPublishedSurface = AgentLaunchResult @@ -109,7 +110,7 @@ export async function executeAgentLaunch( outcome: { kind: 'terminal', handle: intent.reuseTerminal.handle }, worktreeId: existingWorktreeId(intent.target), receipt: preflight, - ...promptReceipt(intent, settledAtCreation({})) + ...promptReceipt(intent, settledAtCreation(intent, {})) }) return { ...reused, @@ -134,7 +135,7 @@ export async function executeAgentLaunch( worktreeId: placed.worktreeId, receipt: preflight, ...(placed.warning ? { warning: placed.warning } : {}), - ...promptReceipt(intent, settledAtCreation(placed)) + ...promptReceipt(intent, settledAtCreation(intent, placed)) }) return { ...startup, @@ -192,7 +193,7 @@ export async function executeAgentLaunch( worktreeId: placed.worktreeId, receipt: settled, ...(warning ? { warning } : {}), - ...promptReceipt(intent, settledAtCreation(created)) + ...promptReceipt(intent, settledAtCreation(intent, created)) }) return { ...surface, diff --git a/src/main/agent-launch/agent-launch-prompt-delivery.ts b/src/main/agent-launch/agent-launch-prompt-delivery.ts index 226c17cff82..a1e062b8cbc 100644 --- a/src/main/agent-launch/agent-launch-prompt-delivery.ts +++ b/src/main/agent-launch/agent-launch-prompt-delivery.ts @@ -9,6 +9,7 @@ * terminal, line fits -> folded into the command that execs the agent -> handed-to-terminal * terminal, otherwise -> bracketed paste into the live PTY once it is ready -> handed-to-terminal * anything unproven -> -> not-delivered + * host stopped mid-delivery (a replayed record only) -> unconfirmed * * argv has no readiness race, so it is offered wherever the agent's CLI takes a prompt argument * (`agentPromptRidesLaunchCommand`). But that command is TYPED into the user's shell, and a long or @@ -28,16 +29,22 @@ import type { AgentLaunchStructuredSurface } from './agent-launch-surface-factor export const HANDED_TO_TERMINAL: AgentLaunchPromptDisposal = { outcome: 'handed-to-terminal' } const NOT_DELIVERED: AgentLaunchPromptDisposal = { outcome: 'not-delivered' } +const UNCONFIRMED: AgentLaunchPromptDisposal = { outcome: 'unconfirmed' } /** - * What the receipt may say before any delivery runs: only what creating the surface already - * settled. A launch command that carried the text has handed it over; anything else is not (yet) - * delivered, which is also the truthful answer when the host dies before delivering it. + * What the record may say before any delivery runs: only what creating the surface already settled. + * A launch command that carried the text has handed it over, and a draft is never delivered by the + * host. A submit still owed is `unconfirmed`: a host that stops mid-delivery cannot say whether the + * paste or the commit landed, and "not delivered" would invite a duplicate turn. */ -export function settledAtCreation(created: { - promptRodeLaunchCommand?: boolean -}): AgentLaunchPromptDisposal { - return created.promptRodeLaunchCommand ? HANDED_TO_TERMINAL : NOT_DELIVERED +export function settledAtCreation( + intent: Pick, + created: { promptRodeLaunchCommand?: boolean } +): AgentLaunchPromptDisposal { + if (created.promptRodeLaunchCommand) { + return HANDED_TO_TERMINAL + } + return intent.prompt?.delivery === 'submit' ? UNCONFIRMED : NOT_DELIVERED } /** Each surface delivers its own way, so the disposal is decided where the surface is known. */ @@ -138,9 +145,9 @@ export function launchCommandPrompt( * Each arm is a consequence of the act it names, never a write-ahead of it: `journaled` is * reachable only from a committed message id, `handed-to-terminal` only from a launch command that * carried the text or a PTY write that returned, and everything else under-claims as - * `not-delivered`. There is deliberately no arm for "maybe" — a caller holding one could neither - * resend nor drop the text. Dispatch doubt is not this tier's to report: the submission row - * carries it. + * `not-delivered`. A live answer has no "maybe": `unconfirmed` is written only into the record + * before delivery runs (`settledAtCreation`), and is read back only by a replay. Dispatch doubt is + * not this tier's to report: the submission row carries it. */ export function promptReceipt( intent: AgentLaunchIntent, diff --git a/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts b/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts index cb70e51b6f9..4f071a383bc 100644 --- a/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts +++ b/src/main/ipc/desktop-renderer-runtime-capabilities.test.ts @@ -26,7 +26,10 @@ import { WORKTREE_VISIBILITY_SOURCE_DEFAULTS_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../shared/protocol-version' -import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability' +import { + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY, + AGENT_LAUNCH_RUNTIME_CAPABILITY +} from '../../shared/agent-launch-runtime-capability' import { ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES } from '../../shared/electron-remote-runtime-client-capabilities' import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability' import { supportsAgentLaunch } from '../runtime/rpc/methods/agent-launch' @@ -58,7 +61,7 @@ const REMOTE_ONLY_BY_DECISION: readonly RuntimeCapability[] = [ ] /** Gates the renderer must pass against its own main process. The Electron remote list omits all - * seven; mobile advertises the structured ones, so this is an Electron-remote gap rather than a + * eight; mobile advertises the structured ones, so this is an Electron-remote gap rather than a * statement that no remote client wants them. Why it is one is not recorded here. */ const LOCAL_ONLY_BY_DECISION: readonly RuntimeCapability[] = [ AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, @@ -68,7 +71,9 @@ const LOCAL_ONLY_BY_DECISION: readonly RuntimeCapability[] = [ STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, // The desktop picks each launch mode itself; only its own host is told so far. - STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY + STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY, + // Read by the desktop's own launches first; a remote host is told when its launches move over. + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY ] function missingFrom( diff --git a/src/main/ipc/desktop-renderer-runtime-capabilities.ts b/src/main/ipc/desktop-renderer-runtime-capabilities.ts index f1eb84dca40..78219b6b09b 100644 --- a/src/main/ipc/desktop-renderer-runtime-capabilities.ts +++ b/src/main/ipc/desktop-renderer-runtime-capabilities.ts @@ -9,7 +9,10 @@ import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../shared/protocol-version' -import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability' +import { + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY, + AGENT_LAUNCH_RUNTIME_CAPABILITY +} from '../../shared/agent-launch-runtime-capability' import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability' /** @@ -37,5 +40,8 @@ export const DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES: readonly RuntimeCapab STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY, // Without this `supportsAgentLaunch` refuses the renderer outright, while the same renderer // targeting a remote host is admitted — the asymmetry this constant exists to close. - AGENT_LAUNCH_RUNTIME_CAPABILITY + AGENT_LAUNCH_RUNTIME_CAPABILITY, + // A replay after a restart mid-delivery answers the running agent with an `unconfirmed` prompt, + // which the desktop reads, rather than refusing it as unknown. + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY ] as const diff --git a/src/main/runtime/agent-session-record-store-slot.ts b/src/main/runtime/agent-session-record-store-slot.ts index a05de7a60ce..3e74f88282c 100644 --- a/src/main/runtime/agent-session-record-store-slot.ts +++ b/src/main/runtime/agent-session-record-store-slot.ts @@ -5,6 +5,11 @@ // whole chat host (provider adapters, the host, the model catalog) first. The store is opened here // instead and the chat host, when something needs it, is built on this same instance: the store is // a single writer, so a second copy of it would diverge. +// +// Known edges, both reachable only once quit has begun (stop has no other production caller): a +// stop whose host teardown fails empties the slot while that host still holds its store open, so a +// later admission would open a second one; and a store opened after stop is closed by nothing but +// process exit. import { AgentSessionRecordStore } from './agent-session-record-store' import type { JournalHostDatabase } from '../native-chat/agent-session-journal/journal-host-database' diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 5bb774af3b1..de3d63b4b96 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -151,8 +151,9 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStartTuiIdleVis /** * Installs the structured agent-session host on first use. Lazy for the same * reason the orchestration DB is: the profile's user-data path is not final - * until the app is ready, and a runtime nobody drives a chat session on - * should never open the record store. + * until the app is ready, and a runtime nobody drives a chat session on never + * builds the chat host. The record store it sits on may already be open, from + * a launch's admission. */ async ensureStructuredAgentSessionHost(): Promise { await installStructuredAgentSessionHost({ diff --git a/src/main/runtime/rpc/methods/agent-launch-replay.ts b/src/main/runtime/rpc/methods/agent-launch-replay.ts index 711a5ebf208..8e16ffdf991 100644 --- a/src/main/runtime/rpc/methods/agent-launch-replay.ts +++ b/src/main/runtime/rpc/methods/agent-launch-replay.ts @@ -16,6 +16,7 @@ */ import { deriveAgentLaunchChildOperationId } from '../../../../shared/agent-launch-operation' +import { AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY } from '../../../../shared/agent-launch-runtime-capability' import { isAgentLaunchResult, type AgentLaunchResult } from '../../../../shared/agent-launch-intent' import type { AgentSessionOperationOutcome, @@ -100,6 +101,60 @@ function answerFromRecordedRow( : { decision: 'refuse', refusal: replay.refusal } } +/** The CLI (no declared client) ships with this host; any other caller must say it reads the word. */ +function readsUnconfirmedLaunchPrompt( + context: Pick +): boolean { + return ( + context.clientKind === undefined || + context.clientCapabilities?.includes(AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY) === + true + ) +} + +/** + * A terminal handle is issued by the process that answered, so a recorded one is dead after a + * restart. The pane key is the durable name: the handle is re-derived from it in this runtime. A + * pane this runtime no longer knows keeps the recorded handle, which then resolves to not-found — + * the truth about a terminal that is gone. The handle stays because shipped clients require one. + */ +function withLiveTerminalHandle( + recorded: AgentLaunchResult, + runtime: Pick +): AgentLaunchResult { + const { outcome } = recorded + if (outcome.kind !== 'terminal' || !outcome.paneKey) { + return recorded + } + const handle = runtime.getTerminalHandleForPaneKey(outcome.paneKey) + return handle && handle !== outcome.handle + ? { ...recorded, outcome: { ...outcome, handle } } + : recorded +} + +/** + * A recorded answer as this caller may read it now. A prompt the host stopped delivering is + * `unconfirmed` only to a caller that reads the word; every other caller gets the refusal it got + * before the first write existed, never a `not-delivered` that invites a duplicate turn. + */ +function presentRecordedAnswer( + context: RpcContext, + operationId: string, + answer: AgentLaunchAdmission +): AgentLaunchAdmission { + if (answer.decision !== 'replay') { + return answer + } + if (answer.result.prompt?.outcome === 'unconfirmed' && !readsUnconfirmedLaunchPrompt(context)) { + return refusal( + operationId, + 'agent_session_operation_unknown', + 'started its agent, but whether its prompt arrived is unknown' + ) + } + return { decision: 'replay', result: withLiveTerminalHandle(answer.result, context.runtime) } +} + /** * Admit, then claim, in one durable transaction so a launch writes the ledger once before its effect. * @@ -136,7 +191,7 @@ export async function admitAgentLaunchOperation( if (admitted.decision === 'replay') { const answer = answerFromRecordedRow(operationId, admitted.row.outcome) if (answer) { - return answer + return presentRecordedAnswer(context, operationId, answer) } } // Unreachable with both steps in one transaction; answered as uncertain rather than run twice. @@ -150,10 +205,10 @@ export async function admitAgentLaunchOperation( if (claim.claim === 'lost') { // The handler joins same-process retries before admission. Reaching a claimed row here means // this runtime did not start it, so treating it as restart uncertainty is the safe answer. - return ( - answerFromRecordedRow(operationId, claim.row.outcome) ?? - refusal(operationId, 'agent_session_operation_unknown', 'is claimed but unsettled') - ) + const answer = answerFromRecordedRow(operationId, claim.row.outcome) + return answer + ? presentRecordedAnswer(context, operationId, answer) + : refusal(operationId, 'agent_session_operation_unknown', 'is claimed but unsettled') } const succeeded = (result: AgentLaunchResult) => store.recordOperationOutcome({ diff --git a/src/main/runtime/rpc/methods/agent-launch-restart-replay.test.ts b/src/main/runtime/rpc/methods/agent-launch-restart-replay.test.ts index 075631fb655..3865afebd90 100644 --- a/src/main/runtime/rpc/methods/agent-launch-restart-replay.test.ts +++ b/src/main/runtime/rpc/methods/agent-launch-restart-replay.test.ts @@ -4,15 +4,19 @@ * The record is written twice: once when the surface exists and once when the prompt's fate is * known. A host that dies at any point leaves the replay a truthful answer — the running agent once * its surface is recorded, an honest "unknown" before — and never a second agent. A "restart" here - * is what a new process sees: the store reopened from disk and a runtime with no in-flight launches. + * is what a new process sees: the store reopened from disk and a runtime with no in-flight launches, + * which issues its own handles, so a surviving pane comes back under a new one. */ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../../../shared/agent-launch-runtime-capability' -import type { AgentLaunchResult } from '../../../../shared/agent-launch-intent' +import { + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY, + AGENT_LAUNCH_RUNTIME_CAPABILITY +} from '../../../../shared/agent-launch-runtime-capability' +import { isAgentLaunchResult, type AgentLaunchResult } from '../../../../shared/agent-launch-intent' import type { AgentSessionOperationRow } from '../../../../shared/agent-session-operation-ledger' import type { AgentSessionRecordStore } from '../../agent-session-record-store' import { openTestAgentSessionRecordStore } from '../../agent-session-record-store-test-harness' @@ -20,6 +24,7 @@ import type { OrcaRuntimeService } from '../../orca-runtime' import type { RpcContext } from '../core' import { RpcDispatcher } from '../dispatcher' import { DESKTOP_RPC_CALLER } from '../rpc-caller-identity' +import { DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES } from '../../../ipc/desktop-renderer-runtime-capabilities' import { methodNamed, rpcContext, @@ -54,22 +59,45 @@ const PHONE: Partial = { pairedDeviceId: 'device-1', clientCapabilities: [AGENT_LAUNCH_RUNTIME_CAPABILITY] } +/** The same paired device on a build that reads an `unconfirmed` prompt. */ +const UPGRADED_PHONE: Partial = { + ...PHONE, + clientCapabilities: [ + AGENT_LAUNCH_RUNTIME_CAPABILITY, + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY + ] +} /** Exactly what the desktop's `runtime:call` handler hands the dispatcher. */ const DESKTOP_IPC = { clientId: 'desktop-renderer', caller: DESKTOP_RPC_CALLER, clientKind: 'runtime' as const, - clientCapabilities: [AGENT_LAUNCH_RUNTIME_CAPABILITY] + clientCapabilities: DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES } +/** The handle a restarted host issues the pane that outlived the old one. */ +const ADOPTED_HANDLE = 'term_adopted' let directory: string let store: AgentSessionRecordStore -function hostRuntime() { +function hostRuntime(adoptedPanes?: Record) { // A phone's launch into an existing workspace also moves the phone's own view to the new tab. - return Object.assign(runtimeStub({ settings: TERMINAL_ONLY, terminalPaneKey: PANE_KEY }), { - selectCreatedMobileSessionTabForClient: vi.fn(() => true) - }) + return Object.assign( + runtimeStub({ settings: TERMINAL_ONLY, terminalPaneKey: PANE_KEY, adoptedPanes }), + { selectCreatedMobileSessionTabForClient: vi.fn(() => true) } + ) +} + +/** The host after a restart, still running the pane the dead one launched. */ +function restartedHostRuntime() { + return hostRuntime({ [PANE_KEY]: ADOPTED_HANDLE }) +} + +const UNCONFIRMED_AGENT = { + outcome: { kind: 'terminal', handle: ADOPTED_HANDLE, paneKey: PANE_KEY }, + worktreeId: 'wt-7', + receipt: expect.objectContaining({ mode: 'terminal' }), + prompt: { delivery: 'submit', outcome: 'unconfirmed' } } function launch( @@ -113,6 +141,46 @@ async function ledgerWritesQueuedBefore(ledger: AgentSessionRecordStore): Promis }) } +/** Launches with a paste that never returns: the host dies while it waits for the agent. */ +async function launchUntilPasteStarts(runtime: AgentLaunchRuntimeStub): Promise { + let pasting: () => void = () => {} + const pasteStarted = new Promise((resolve) => { + pasting = resolve + }) + deliverTerminalPrompt.mockImplementationOnce(() => { + pasting() + return new Promise(() => {}) + }) + void launch(runtime) + await pasteStarted + await ledgerWritesQueuedBefore(store) +} + +/** The ledger as the launch reaches it, with the launch's `nth` write (1-based) replaced. */ +function replaceLaunchWrite( + ledger: AgentSessionRecordStore, + nth: number, + write: () => Promise +): void { + const recordOperationOutcome = ledger.recordOperationOutcome.bind(ledger) + let launchWrites = 0 + vi.spyOn(ledger, 'recordOperationOutcome').mockImplementation((input) => { + if (input.operationId === OPERATION_ID && input.callerKey === 'device-1') { + launchWrites += 1 + if (launchWrites === nth) { + return write() + } + } + return recordOperationOutcome(input) + }) +} + +function promptOutcome(outcome: AgentSessionOperationRow['outcome'] | undefined): unknown { + return outcome?.status === 'succeeded' && isAgentLaunchResult(outcome.launch) + ? outcome.launch.prompt?.outcome + : undefined +} + function dispatcherFor(runtime: AgentLaunchRuntimeStub): RpcDispatcher { return new RpcDispatcher({ // oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: the fixture implements every runtime method agent.launch reaches, plus the id the dispatcher stamps on replies. @@ -135,37 +203,73 @@ afterEach(async () => { }) describe('a host restart mid-launch', () => { - it('finds the running agent when the host died while its prompt waited for readiness', async () => { - // The paste waits for the agent forever: the host dies first. - let waiting: () => void = () => {} - const readinessWait = new Promise((resolve) => { - waiting = resolve - }) - deliverTerminalPrompt.mockImplementationOnce(() => { - waiting() - return new Promise(() => {}) - }) + it('finds the running agent, under the handle the new host issued, after a restart mid-paste', async () => { const dying = hostRuntime() - void launch(dying) - await readinessWait - await ledgerWritesQueuedBefore(store) + await launchUntilPasteStarts(dying) await restartHost() - const restarted = hostRuntime() - const replayed = await launch(restarted) + const restarted = restartedHostRuntime() - expect(replayed).toEqual({ - outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY }, - worktreeId: 'wt-7', - receipt: expect.objectContaining({ mode: 'terminal' }), - // The dead host never pasted it, and saying so lets the caller keep the text. - prompt: { delivery: 'submit', outcome: 'not-delivered' } - }) + // The dead host's `term_1` names nothing now; the pane is the durable name. + await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual( + UNCONFIRMED_AGENT + ) expect(restarted.createTerminal).not.toHaveBeenCalled() expect(dying.createTerminal).toHaveBeenCalledOnce() expect(deliverTerminalPrompt).toHaveBeenCalledOnce() }) + it('answers a caller that cannot read "unconfirmed" as before: unknown, never "not sent"', async () => { + await launchUntilPasteStarts(hostRuntime()) + + await restartHost() + const restarted = restartedHostRuntime() + + await expect(launch(restarted)).rejects.toThrow('agent_session_operation_unknown') + // The same row still answers a caller that reads the word: the refusal is presentation only. + await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual( + UNCONFIRMED_AGENT + ) + expect(restarted.createTerminal).not.toHaveBeenCalled() + expect(deliverTerminalPrompt).toHaveBeenCalledOnce() + }) + + it('stays unconfirmed when the host died after the paste returned, before the final write', async () => { + let pasted: () => void = () => {} + const pasteReturned = new Promise((resolve) => { + pasted = resolve + }) + deliverTerminalPrompt.mockImplementationOnce(async () => { + pasted() + return true + }) + // The final write never lands: the process is gone by then. + replaceLaunchWrite(store, 2, () => new Promise(() => {})) + void launch(hostRuntime()) + await pasteReturned + await untilRecorded('succeeded') + await ledgerWritesQueuedBefore(store) + + await restartHost() + const restarted = restartedHostRuntime() + + await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual( + UNCONFIRMED_AGENT + ) + await expect(launch(restarted)).rejects.toThrow('agent_session_operation_unknown') + expect(deliverTerminalPrompt).toHaveBeenCalledOnce() + }) + + it('keeps the recorded handle when the restarted host no longer has the pane', async () => { + await launchUntilPasteStarts(hostRuntime()) + + await restartHost() + + await expect(launch(hostRuntime(), PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toMatchObject({ + outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY } + }) + }) + it('stays unknown when the host died before the surface was recorded', async () => { const dying = hostRuntime() let spawnRequested: () => void = () => {} @@ -193,9 +297,12 @@ describe('a host restart mid-launch', () => { expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) await restartHost() - const restarted = hostRuntime() + const restarted = restartedHostRuntime() - await expect(launch(restarted)).resolves.toEqual(first) + await expect(launch(restarted)).resolves.toEqual({ + ...first, + outcome: { ...first.outcome, handle: ADOPTED_HANDLE } + }) expect(restarted.createTerminal).not.toHaveBeenCalled() expect(deliverTerminalPrompt).toHaveBeenCalledOnce() }) @@ -221,6 +328,50 @@ describe('a host restart mid-launch', () => { }) }) +describe('a final write that fails', () => { + it('leaves the prompt unconfirmed, or unknown to a caller that cannot read it, never "not sent"', async () => { + replaceLaunchWrite(store, 2, () => Promise.reject(new Error('SQLITE_BUSY'))) + const host = hostRuntime() + + const first = await launch(host) + + // The live answer is the truth the host saw; only the record missed it. + expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) + await expect(launch(host)).rejects.toThrow('agent_session_operation_unknown') + await expect(launch(host, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toMatchObject({ + outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY }, + prompt: { delivery: 'submit', outcome: 'unconfirmed' } + }) + expect(host.createTerminal).toHaveBeenCalledOnce() + expect(deliverTerminalPrompt).toHaveBeenCalledOnce() + }) +}) + +describe('the two writes', () => { + it('land in order when the paste returns before the first write has committed', async () => { + const recorded = vi.spyOn(store, 'recordOperationOutcome') + + await launch(hostRuntime()) + + // Both were queued before either committed; the second is the one left standing. + const launchWrites = recorded.mock.calls.filter(([input]) => input.callerKey === 'device-1') + expect(launchWrites.map(([input]) => promptOutcome(input.outcome))).toEqual([ + 'unconfirmed', + 'handed-to-terminal' + ]) + expect(promptOutcome(row()?.outcome)).toBe('handed-to-terminal') + }) + + it('never lets a failed first write block the launch', async () => { + replaceLaunchWrite(store, 1, () => Promise.reject(new Error('SQLITE_BUSY'))) + + const first = await launch(hostRuntime()) + + expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' }) + expect(promptOutcome(row()?.outcome)).toBe('handed-to-terminal') + }) +}) + describe('a reply lost three times', () => { it('starts one agent, and every retry gets its answer', async () => { const host = hostRuntime() @@ -276,6 +427,30 @@ describe('the desktop launches replay-safely', () => { expect(row('trusted-local:desktop')?.outcome.status).toBe('succeeded') }) + it('reads an unconfirmed prompt after a restart mid-paste, through its own IPC transport', async () => { + const request = { + id: 'request-1', + authToken: 'desktop-ipc', + method: 'agent.launchReplay', + params: PROMPTED_LAUNCH + } + deliverTerminalPrompt.mockImplementationOnce(() => new Promise(() => {})) + void dispatcherFor(hostRuntime()).dispatch(request, DESKTOP_IPC) + const deadline = Date.now() + 2_000 + while (row('trusted-local:desktop')?.outcome.status !== 'succeeded') { + if (Date.now() > deadline) { + throw new Error('the desktop launch was never recorded') + } + await new Promise((resolve) => setTimeout(resolve, 5)) + } + await ledgerWritesQueuedBefore(store) + + await restartHost() + const replayed = await dispatcherFor(restartedHostRuntime()).dispatch(request, DESKTOP_IPC) + + expect(replayed).toMatchObject({ ok: true, result: UNCONFIRMED_AGENT }) + }) + it('opens the ledger alone, never the chat host, to admit a terminal launch', async () => { const host = hostRuntime() diff --git a/src/main/runtime/rpc/methods/agent-launch.test-fixture.ts b/src/main/runtime/rpc/methods/agent-launch.test-fixture.ts index 13122f49e57..dcd6e0e6bda 100644 --- a/src/main/runtime/rpc/methods/agent-launch.test-fixture.ts +++ b/src/main/runtime/rpc/methods/agent-launch.test-fixture.ts @@ -39,6 +39,9 @@ export type AgentLaunchRuntimeStubOptions = { terminalPaneAlreadyLive?: boolean /** What the runtime reports about an offered prompt's typed line; unset reports nothing. */ lineCarriesPrompt?: boolean + /** Panes this runtime found already running, by the handle it issued them: a restarted host + * adopting a surviving PTY issues a new handle for the same pane. */ + adoptedPanes?: Record } function reportPromptCarry( @@ -60,6 +63,8 @@ export function setAgentLaunchRecordStore(store: AgentSessionRecordStore | null) export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) { const worktreeCreateResults = new Map>() + // Only the panes this runtime created or adopted: a handle is process-scoped, a pane key is not. + const handlesByPaneKey = new Map(Object.entries(options.adoptedPanes ?? {})) const waitForSetupTerminalCompletion = vi.fn( async (_handle: string, _signal?: AbortSignal): Promise<{ exitCode: number | null }> => ({ exitCode: 0 @@ -89,6 +94,9 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) { showRepo: vi.fn(async () => ({ id: 'repo-1' })), createManagedWorktree: vi.fn(async (args: Record) => { reportPromptCarry(options, args.onStartupPromptCarry, args.startupPrompt) + if (args.startupAgent && options.startupTerminalPaneKey) { + handlesByPaneKey.set(options.startupTerminalPaneKey, 'term_agent_first') + } return { worktree: { id: 'wt-new' }, startupTerminal: args.startupAgent @@ -107,6 +115,9 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) { throw new AgentLaunchPaneAlreadyLiveError() } reportPromptCarry(options, createOptions?.onStartupPromptCarry, createOptions?.startupPrompt) + if (options.terminalPaneKey) { + handlesByPaneKey.set(options.terminalPaneKey, 'term_1') + } return { handle: 'term_1', ...(options.terminalPaneKey ? { paneKey: options.terminalPaneKey } : {}), @@ -114,6 +125,7 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) { } }), showTerminal: vi.fn(async (handle: string) => ({ handle, worktreeId: 'wt-7' })), + getTerminalHandleForPaneKey: vi.fn((paneKey: string) => handlesByPaneKey.get(paneKey) ?? null), isTerminalRunningAgent: vi.fn(async () => true), showManagedTerminalWorkspace: vi.fn(async (selector: string) => ({ id: selector.replace(/^id:/, '') diff --git a/src/main/runtime/rpc/methods/agent-launch.ts b/src/main/runtime/rpc/methods/agent-launch.ts index 81eafd1d4d5..61424c2d5dc 100644 --- a/src/main/runtime/rpc/methods/agent-launch.ts +++ b/src/main/runtime/rpc/methods/agent-launch.ts @@ -147,8 +147,9 @@ type ReplaySafeLaunch = { attachOperationId: string callerKey: string terminalSpawn: TerminalSpawnDispatch - /** Records the surface the moment it exists. Fired, never awaited: the ledger's transactions run - * in order, so the final settle still lands after it, and the prompt never waits on bookkeeping. */ + /** Records the surface the moment it exists, an owed prompt as `unconfirmed`. Fired, never + * awaited: the ledger's transactions run in order, so the final settle still lands after it, and + * the prompt never waits on bookkeeping. */ recordSurface: (provisional: AgentLaunchResult) => void } @@ -280,7 +281,8 @@ async function executeReplaySafeAgentLaunch( } throw new AgentLaunchExecutionError(error, failedWithoutEffects !== null) } - // Settlement is bookkeeping; failure leaves the truthful `unknown` refusal for later retries. + // Bookkeeping: a failure leaves the first write, whose owed prompt replays as `unconfirmed` (or as + // `unknown` to a caller that cannot read it), never as `not-delivered`. await settleQuietly(admission.settle(result)) return result } diff --git a/src/main/runtime/rpc/rpc-caller-identity.ts b/src/main/runtime/rpc/rpc-caller-identity.ts index 826789d40b0..13e5f15e6aa 100644 --- a/src/main/runtime/rpc/rpc-caller-identity.ts +++ b/src/main/runtime/rpc/rpc-caller-identity.ts @@ -24,6 +24,11 @@ export const DESKTOP_RPC_CALLER: RpcCallerIdentity = { kind: 'desktop' } * A transport that names its caller is believed; one that declares no client at all is the * in-process runtime socket, trusted as it always was; one that declares a client it cannot name has * no identity, and anything that needs one refuses it. + * + * Temporary: "no declared client" means the local CLI, so the SSH remote CLI bridge and browser + * automation share its namespace, and a new transport that forgets to stamp its caller inherits it. + * Before plugins ship (plan §4) the runtime socket stamps `local-cli` itself and no stamp means no + * identity. */ export function resolveRpcCallerIdentity( transport: diff --git a/src/shared/agent-launch-intent.ts b/src/shared/agent-launch-intent.ts index bffb93c1b9a..a57d44c82b0 100644 --- a/src/shared/agent-launch-intent.ts +++ b/src/shared/agent-launch-intent.ts @@ -154,6 +154,13 @@ export type AgentLaunchPromptDisposal = | { outcome: 'handed-to-terminal' } /** Not delivered by this call; the caller still owns the text. */ | { outcome: 'not-delivered' } + /** + * Only ever replayed, never a live answer: the host recorded the running agent, then stopped + * before the delivery reported back, so the text may or may not have arrived. The caller must not + * resend. Sent only to a caller advertising `agent.launch.prompt-unconfirmed.v1`; every other + * caller is refused with `agent_session_operation_unknown` instead. + */ + | { outcome: 'unconfirmed' } export type AgentLaunchPromptReceipt = { delivery: AgentLaunchPromptDelivery @@ -242,7 +249,9 @@ function isAgentLaunchPromptReceipt(value: unknown): value is AgentLaunchPromptR } return value.outcome === 'journaled' ? 'messageId' in value && typeof value.messageId === 'string' - : value.outcome === 'handed-to-terminal' || value.outcome === 'not-delivered' + : value.outcome === 'handed-to-terminal' || + value.outcome === 'not-delivered' || + value.outcome === 'unconfirmed' } function isAgentLaunchOutcome(value: unknown): value is AgentLaunchOutcome { diff --git a/src/shared/agent-launch-runtime-capability.ts b/src/shared/agent-launch-runtime-capability.ts index 2094f705dc7..e4853e84f87 100644 --- a/src/shared/agent-launch-runtime-capability.ts +++ b/src/shared/agent-launch-runtime-capability.ts @@ -29,9 +29,15 @@ export const AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY = 'agent.launch.prompt export const AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY = 'agent.launch.replay-required.v1' as const +// A client that reads `prompt.outcome: 'unconfirmed'` on a replayed launch. Without it, a launch +// whose host stopped mid-delivery replays as `agent_session_operation_unknown`, as it always has. +export const AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY = + 'agent.launch.prompt-unconfirmed.v1' as const + export const AGENT_LAUNCH_RUNTIME_CAPABILITIES = [ AGENT_LAUNCH_RUNTIME_CAPABILITY, AGENT_LAUNCH_REPLAY_RUNTIME_CAPABILITY, AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY, - AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY + AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY, + AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY ] as const