fix(agent-launch): a replayed launch names its live terminal and never calls an unfinished prompt unsent

Two answers a retried launch got after a host restart were wrong.

The recorded terminal handle belonged to the process that issued it, so after
a restart it named nothing, even though the pane was still running. A replay
now re-derives the handle from the recorded pane key in the current runtime.
A pane that is gone keeps the recorded handle, which resolves to not-found;
shipped clients require a handle, so it cannot be dropped.

The record written when the tab appears said the prompt was "not delivered",
and a failed final write left that standing. A host that stopped mid-paste
cannot know that, and "not delivered" invites a resend that becomes a
duplicate turn. The first write now records a prompt still owed as
"unconfirmed". On replay, a caller that advertises
agent.launch.prompt-unconfirmed.v1 (the desktop, and the CLI, which ships with
the host) gets the running agent with that word; every other caller gets
agent_session_operation_unknown, the answer it got before the first write
existed. An older build reading the row rejects the word and answers the same.
This commit is contained in:
Brennan Benson
2026-10-04 14:07:44 -07:00
parent 0d7d9b5b30
commit c83f44dd72
14 changed files with 369 additions and 67 deletions
@@ -615,7 +615,7 @@ describe('the surface is published as the launch stands, before its prompt is de
return { launch, published }
}
it('records a prompt still owed as not delivered, then delivers it', async () => {
it('records a prompt still owed as unconfirmed, then delivers it', async () => {
const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: false })
const result = await launch.run(PROMPTED_EXISTING)
@@ -626,12 +626,25 @@ describe('the surface is published as the launch stands, before its prompt is de
outcome: { kind: 'terminal', handle: 'term_1' },
worktreeId: 'wt-7',
receipt: result.receipt,
prompt: { delivery: 'submit', outcome: 'not-delivered' }
// A host that stops mid-paste cannot say whether it landed, so it must not say "not sent".
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
}
])
expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
})
it('records a draft as not delivered, since the host never delivers one', async () => {
const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: false })
const result = await launch.run({
...PROMPTED_EXISTING,
prompt: { text: 'fix the build', delivery: 'draft' }
})
expect(published[0]?.prompt).toEqual({ delivery: 'draft', outcome: 'not-delivered' })
expect(result.prompt).toEqual({ delivery: 'draft', outcome: 'not-delivered' })
})
it('records a prompt the launch command carried as already handed over', async () => {
const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: true })
@@ -655,7 +668,7 @@ describe('the surface is published as the launch stands, before its prompt is de
])
expect(published[0]).toEqual({
...result,
prompt: { delivery: 'submit', outcome: 'not-delivered' }
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
})
expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'journaled', messageId: 'msg-1' })
})
@@ -77,8 +77,9 @@ export type AgentLaunchExecution = {
/**
* The launch as it stands once its surface exists: a complete result whose prompt receipt says only
* what creation itself settled — carried on the launch command, or not (yet) delivered. Complete so
* a host that dies during the delivery still leaves a truthful answer behind.
* what creation itself settled — carried on the launch command, a draft the host never delivers, or
* a submit still `unconfirmed`. Complete so a host that dies during the delivery still leaves a
* truthful answer behind.
*/
export type AgentLaunchPublishedSurface = AgentLaunchResult
@@ -109,7 +110,7 @@ export async function executeAgentLaunch(
outcome: { kind: 'terminal', handle: intent.reuseTerminal.handle },
worktreeId: existingWorktreeId(intent.target),
receipt: preflight,
...promptReceipt(intent, settledAtCreation({}))
...promptReceipt(intent, settledAtCreation(intent, {}))
})
return {
...reused,
@@ -134,7 +135,7 @@ export async function executeAgentLaunch(
worktreeId: placed.worktreeId,
receipt: preflight,
...(placed.warning ? { warning: placed.warning } : {}),
...promptReceipt(intent, settledAtCreation(placed))
...promptReceipt(intent, settledAtCreation(intent, placed))
})
return {
...startup,
@@ -192,7 +193,7 @@ export async function executeAgentLaunch(
worktreeId: placed.worktreeId,
receipt: settled,
...(warning ? { warning } : {}),
...promptReceipt(intent, settledAtCreation(created))
...promptReceipt(intent, settledAtCreation(intent, created))
})
return {
...surface,
@@ -9,6 +9,7 @@
* terminal, line fits -> folded into the command that execs the agent -> handed-to-terminal
* terminal, otherwise -> bracketed paste into the live PTY once it is ready -> handed-to-terminal
* anything unproven -> -> not-delivered
* host stopped mid-delivery (a replayed record only) -> unconfirmed
*
* argv has no readiness race, so it is offered wherever the agent's CLI takes a prompt argument
* (`agentPromptRidesLaunchCommand`). But that command is TYPED into the user's shell, and a long or
@@ -28,16 +29,22 @@ import type { AgentLaunchStructuredSurface } from './agent-launch-surface-factor
export const HANDED_TO_TERMINAL: AgentLaunchPromptDisposal = { outcome: 'handed-to-terminal' }
const NOT_DELIVERED: AgentLaunchPromptDisposal = { outcome: 'not-delivered' }
const UNCONFIRMED: AgentLaunchPromptDisposal = { outcome: 'unconfirmed' }
/**
* What the receipt may say before any delivery runs: only what creating the surface already
* settled. A launch command that carried the text has handed it over; anything else is not (yet)
* delivered, which is also the truthful answer when the host dies before delivering it.
* What the record may say before any delivery runs: only what creating the surface already settled.
* A launch command that carried the text has handed it over, and a draft is never delivered by the
* host. A submit still owed is `unconfirmed`: a host that stops mid-delivery cannot say whether the
* paste or the commit landed, and "not delivered" would invite a duplicate turn.
*/
export function settledAtCreation(created: {
promptRodeLaunchCommand?: boolean
}): AgentLaunchPromptDisposal {
return created.promptRodeLaunchCommand ? HANDED_TO_TERMINAL : NOT_DELIVERED
export function settledAtCreation(
intent: Pick<AgentLaunchIntent, 'prompt'>,
created: { promptRodeLaunchCommand?: boolean }
): AgentLaunchPromptDisposal {
if (created.promptRodeLaunchCommand) {
return HANDED_TO_TERMINAL
}
return intent.prompt?.delivery === 'submit' ? UNCONFIRMED : NOT_DELIVERED
}
/** Each surface delivers its own way, so the disposal is decided where the surface is known. */
@@ -138,9 +145,9 @@ export function launchCommandPrompt(
* Each arm is a consequence of the act it names, never a write-ahead of it: `journaled` is
* reachable only from a committed message id, `handed-to-terminal` only from a launch command that
* carried the text or a PTY write that returned, and everything else under-claims as
* `not-delivered`. There is deliberately no arm for "maybe" — a caller holding one could neither
* resend nor drop the text. Dispatch doubt is not this tier's to report: the submission row
* carries it.
* `not-delivered`. A live answer has no "maybe": `unconfirmed` is written only into the record
* before delivery runs (`settledAtCreation`), and is read back only by a replay. Dispatch doubt is
* not this tier's to report: the submission row carries it.
*/
export function promptReceipt(
intent: AgentLaunchIntent,
@@ -26,7 +26,10 @@ import {
WORKTREE_VISIBILITY_SOURCE_DEFAULTS_RUNTIME_CAPABILITY,
type RuntimeCapability
} from '../../shared/protocol-version'
import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability'
import {
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY,
AGENT_LAUNCH_RUNTIME_CAPABILITY
} from '../../shared/agent-launch-runtime-capability'
import { ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES } from '../../shared/electron-remote-runtime-client-capabilities'
import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability'
import { supportsAgentLaunch } from '../runtime/rpc/methods/agent-launch'
@@ -58,7 +61,7 @@ const REMOTE_ONLY_BY_DECISION: readonly RuntimeCapability[] = [
]
/** Gates the renderer must pass against its own main process. The Electron remote list omits all
* seven; mobile advertises the structured ones, so this is an Electron-remote gap rather than a
* eight; mobile advertises the structured ones, so this is an Electron-remote gap rather than a
* statement that no remote client wants them. Why it is one is not recorded here. */
const LOCAL_ONLY_BY_DECISION: readonly RuntimeCapability[] = [
AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY,
@@ -68,7 +71,9 @@ const LOCAL_ONLY_BY_DECISION: readonly RuntimeCapability[] = [
STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY,
CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY,
// The desktop picks each launch mode itself; only its own host is told so far.
STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY
STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY,
// Read by the desktop's own launches first; a remote host is told when its launches move over.
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
]
function missingFrom(
@@ -9,7 +9,10 @@ import {
STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY,
type RuntimeCapability
} from '../../shared/protocol-version'
import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability'
import {
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY,
AGENT_LAUNCH_RUNTIME_CAPABILITY
} from '../../shared/agent-launch-runtime-capability'
import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability'
/**
@@ -37,5 +40,8 @@ export const DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES: readonly RuntimeCapab
STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY,
// Without this `supportsAgentLaunch` refuses the renderer outright, while the same renderer
// targeting a remote host is admitted — the asymmetry this constant exists to close.
AGENT_LAUNCH_RUNTIME_CAPABILITY
AGENT_LAUNCH_RUNTIME_CAPABILITY,
// A replay after a restart mid-delivery answers the running agent with an `unconfirmed` prompt,
// which the desktop reads, rather than refusing it as unknown.
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
] as const
@@ -5,6 +5,11 @@
// whole chat host (provider adapters, the host, the model catalog) first. The store is opened here
// instead and the chat host, when something needs it, is built on this same instance: the store is
// a single writer, so a second copy of it would diverge.
//
// Known edges, both reachable only once quit has begun (stop has no other production caller): a
// stop whose host teardown fails empties the slot while that host still holds its store open, so a
// later admission would open a second one; and a store opened after stop is closed by nothing but
// process exit.
import { AgentSessionRecordStore } from './agent-session-record-store'
import type { JournalHostDatabase } from '../native-chat/agent-session-journal/journal-host-database'
@@ -151,8 +151,9 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStartTuiIdleVis
/**
* Installs the structured agent-session host on first use. Lazy for the same
* reason the orchestration DB is: the profile's user-data path is not final
* until the app is ready, and a runtime nobody drives a chat session on
* should never open the record store.
* until the app is ready, and a runtime nobody drives a chat session on never
* builds the chat host. The record store it sits on may already be open, from
* a launch's admission.
*/
async ensureStructuredAgentSessionHost(): Promise<void> {
await installStructuredAgentSessionHost({
@@ -16,6 +16,7 @@
*/
import { deriveAgentLaunchChildOperationId } from '../../../../shared/agent-launch-operation'
import { AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY } from '../../../../shared/agent-launch-runtime-capability'
import { isAgentLaunchResult, type AgentLaunchResult } from '../../../../shared/agent-launch-intent'
import type {
AgentSessionOperationOutcome,
@@ -100,6 +101,60 @@ function answerFromRecordedRow(
: { decision: 'refuse', refusal: replay.refusal }
}
/** The CLI (no declared client) ships with this host; any other caller must say it reads the word. */
function readsUnconfirmedLaunchPrompt(
context: Pick<RpcContext, 'clientKind' | 'clientCapabilities'>
): boolean {
return (
context.clientKind === undefined ||
context.clientCapabilities?.includes(AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY) ===
true
)
}
/**
* A terminal handle is issued by the process that answered, so a recorded one is dead after a
* restart. The pane key is the durable name: the handle is re-derived from it in this runtime. A
* pane this runtime no longer knows keeps the recorded handle, which then resolves to not-found —
* the truth about a terminal that is gone. The handle stays because shipped clients require one.
*/
function withLiveTerminalHandle(
recorded: AgentLaunchResult,
runtime: Pick<RpcContext['runtime'], 'getTerminalHandleForPaneKey'>
): AgentLaunchResult {
const { outcome } = recorded
if (outcome.kind !== 'terminal' || !outcome.paneKey) {
return recorded
}
const handle = runtime.getTerminalHandleForPaneKey(outcome.paneKey)
return handle && handle !== outcome.handle
? { ...recorded, outcome: { ...outcome, handle } }
: recorded
}
/**
* A recorded answer as this caller may read it now. A prompt the host stopped delivering is
* `unconfirmed` only to a caller that reads the word; every other caller gets the refusal it got
* before the first write existed, never a `not-delivered` that invites a duplicate turn.
*/
function presentRecordedAnswer(
context: RpcContext,
operationId: string,
answer: AgentLaunchAdmission
): AgentLaunchAdmission {
if (answer.decision !== 'replay') {
return answer
}
if (answer.result.prompt?.outcome === 'unconfirmed' && !readsUnconfirmedLaunchPrompt(context)) {
return refusal(
operationId,
'agent_session_operation_unknown',
'started its agent, but whether its prompt arrived is unknown'
)
}
return { decision: 'replay', result: withLiveTerminalHandle(answer.result, context.runtime) }
}
/**
* Admit, then claim, in one durable transaction so a launch writes the ledger once before its effect.
*
@@ -136,7 +191,7 @@ export async function admitAgentLaunchOperation(
if (admitted.decision === 'replay') {
const answer = answerFromRecordedRow(operationId, admitted.row.outcome)
if (answer) {
return answer
return presentRecordedAnswer(context, operationId, answer)
}
}
// Unreachable with both steps in one transaction; answered as uncertain rather than run twice.
@@ -150,10 +205,10 @@ export async function admitAgentLaunchOperation(
if (claim.claim === 'lost') {
// The handler joins same-process retries before admission. Reaching a claimed row here means
// this runtime did not start it, so treating it as restart uncertainty is the safe answer.
return (
answerFromRecordedRow(operationId, claim.row.outcome) ??
refusal(operationId, 'agent_session_operation_unknown', 'is claimed but unsettled')
)
const answer = answerFromRecordedRow(operationId, claim.row.outcome)
return answer
? presentRecordedAnswer(context, operationId, answer)
: refusal(operationId, 'agent_session_operation_unknown', 'is claimed but unsettled')
}
const succeeded = (result: AgentLaunchResult) =>
store.recordOperationOutcome({
@@ -4,15 +4,19 @@
* The record is written twice: once when the surface exists and once when the prompt's fate is
* known. A host that dies at any point leaves the replay a truthful answer — the running agent once
* its surface is recorded, an honest "unknown" before — and never a second agent. A "restart" here
* is what a new process sees: the store reopened from disk and a runtime with no in-flight launches.
* is what a new process sees: the store reopened from disk and a runtime with no in-flight launches,
* which issues its own handles, so a surviving pane comes back under a new one.
*/
import { mkdtemp, rm } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../../../shared/agent-launch-runtime-capability'
import type { AgentLaunchResult } from '../../../../shared/agent-launch-intent'
import {
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY,
AGENT_LAUNCH_RUNTIME_CAPABILITY
} from '../../../../shared/agent-launch-runtime-capability'
import { isAgentLaunchResult, type AgentLaunchResult } from '../../../../shared/agent-launch-intent'
import type { AgentSessionOperationRow } from '../../../../shared/agent-session-operation-ledger'
import type { AgentSessionRecordStore } from '../../agent-session-record-store'
import { openTestAgentSessionRecordStore } from '../../agent-session-record-store-test-harness'
@@ -20,6 +24,7 @@ import type { OrcaRuntimeService } from '../../orca-runtime'
import type { RpcContext } from '../core'
import { RpcDispatcher } from '../dispatcher'
import { DESKTOP_RPC_CALLER } from '../rpc-caller-identity'
import { DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES } from '../../../ipc/desktop-renderer-runtime-capabilities'
import {
methodNamed,
rpcContext,
@@ -54,22 +59,45 @@ const PHONE: Partial<RpcContext> = {
pairedDeviceId: 'device-1',
clientCapabilities: [AGENT_LAUNCH_RUNTIME_CAPABILITY]
}
/** The same paired device on a build that reads an `unconfirmed` prompt. */
const UPGRADED_PHONE: Partial<RpcContext> = {
...PHONE,
clientCapabilities: [
AGENT_LAUNCH_RUNTIME_CAPABILITY,
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
]
}
/** Exactly what the desktop's `runtime:call` handler hands the dispatcher. */
const DESKTOP_IPC = {
clientId: 'desktop-renderer',
caller: DESKTOP_RPC_CALLER,
clientKind: 'runtime' as const,
clientCapabilities: [AGENT_LAUNCH_RUNTIME_CAPABILITY]
clientCapabilities: DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES
}
/** The handle a restarted host issues the pane that outlived the old one. */
const ADOPTED_HANDLE = 'term_adopted'
let directory: string
let store: AgentSessionRecordStore
function hostRuntime() {
function hostRuntime(adoptedPanes?: Record<string, string>) {
// A phone's launch into an existing workspace also moves the phone's own view to the new tab.
return Object.assign(runtimeStub({ settings: TERMINAL_ONLY, terminalPaneKey: PANE_KEY }), {
selectCreatedMobileSessionTabForClient: vi.fn(() => true)
})
return Object.assign(
runtimeStub({ settings: TERMINAL_ONLY, terminalPaneKey: PANE_KEY, adoptedPanes }),
{ selectCreatedMobileSessionTabForClient: vi.fn(() => true) }
)
}
/** The host after a restart, still running the pane the dead one launched. */
function restartedHostRuntime() {
return hostRuntime({ [PANE_KEY]: ADOPTED_HANDLE })
}
const UNCONFIRMED_AGENT = {
outcome: { kind: 'terminal', handle: ADOPTED_HANDLE, paneKey: PANE_KEY },
worktreeId: 'wt-7',
receipt: expect.objectContaining({ mode: 'terminal' }),
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
}
function launch(
@@ -113,6 +141,46 @@ async function ledgerWritesQueuedBefore(ledger: AgentSessionRecordStore): Promis
})
}
/** Launches with a paste that never returns: the host dies while it waits for the agent. */
async function launchUntilPasteStarts(runtime: AgentLaunchRuntimeStub): Promise<void> {
let pasting: () => void = () => {}
const pasteStarted = new Promise<void>((resolve) => {
pasting = resolve
})
deliverTerminalPrompt.mockImplementationOnce(() => {
pasting()
return new Promise<boolean>(() => {})
})
void launch(runtime)
await pasteStarted
await ledgerWritesQueuedBefore(store)
}
/** The ledger as the launch reaches it, with the launch's `nth` write (1-based) replaced. */
function replaceLaunchWrite(
ledger: AgentSessionRecordStore,
nth: number,
write: () => Promise<void>
): void {
const recordOperationOutcome = ledger.recordOperationOutcome.bind(ledger)
let launchWrites = 0
vi.spyOn(ledger, 'recordOperationOutcome').mockImplementation((input) => {
if (input.operationId === OPERATION_ID && input.callerKey === 'device-1') {
launchWrites += 1
if (launchWrites === nth) {
return write()
}
}
return recordOperationOutcome(input)
})
}
function promptOutcome(outcome: AgentSessionOperationRow['outcome'] | undefined): unknown {
return outcome?.status === 'succeeded' && isAgentLaunchResult(outcome.launch)
? outcome.launch.prompt?.outcome
: undefined
}
function dispatcherFor(runtime: AgentLaunchRuntimeStub): RpcDispatcher {
return new RpcDispatcher({
// oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: the fixture implements every runtime method agent.launch reaches, plus the id the dispatcher stamps on replies.
@@ -135,37 +203,73 @@ afterEach(async () => {
})
describe('a host restart mid-launch', () => {
it('finds the running agent when the host died while its prompt waited for readiness', async () => {
// The paste waits for the agent forever: the host dies first.
let waiting: () => void = () => {}
const readinessWait = new Promise<void>((resolve) => {
waiting = resolve
})
deliverTerminalPrompt.mockImplementationOnce(() => {
waiting()
return new Promise<boolean>(() => {})
})
it('finds the running agent, under the handle the new host issued, after a restart mid-paste', async () => {
const dying = hostRuntime()
void launch(dying)
await readinessWait
await ledgerWritesQueuedBefore(store)
await launchUntilPasteStarts(dying)
await restartHost()
const restarted = hostRuntime()
const replayed = await launch(restarted)
const restarted = restartedHostRuntime()
expect(replayed).toEqual({
outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY },
worktreeId: 'wt-7',
receipt: expect.objectContaining({ mode: 'terminal' }),
// The dead host never pasted it, and saying so lets the caller keep the text.
prompt: { delivery: 'submit', outcome: 'not-delivered' }
})
// The dead host's `term_1` names nothing now; the pane is the durable name.
await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual(
UNCONFIRMED_AGENT
)
expect(restarted.createTerminal).not.toHaveBeenCalled()
expect(dying.createTerminal).toHaveBeenCalledOnce()
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
})
it('answers a caller that cannot read "unconfirmed" as before: unknown, never "not sent"', async () => {
await launchUntilPasteStarts(hostRuntime())
await restartHost()
const restarted = restartedHostRuntime()
await expect(launch(restarted)).rejects.toThrow('agent_session_operation_unknown')
// The same row still answers a caller that reads the word: the refusal is presentation only.
await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual(
UNCONFIRMED_AGENT
)
expect(restarted.createTerminal).not.toHaveBeenCalled()
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
})
it('stays unconfirmed when the host died after the paste returned, before the final write', async () => {
let pasted: () => void = () => {}
const pasteReturned = new Promise<void>((resolve) => {
pasted = resolve
})
deliverTerminalPrompt.mockImplementationOnce(async () => {
pasted()
return true
})
// The final write never lands: the process is gone by then.
replaceLaunchWrite(store, 2, () => new Promise<void>(() => {}))
void launch(hostRuntime())
await pasteReturned
await untilRecorded('succeeded')
await ledgerWritesQueuedBefore(store)
await restartHost()
const restarted = restartedHostRuntime()
await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual(
UNCONFIRMED_AGENT
)
await expect(launch(restarted)).rejects.toThrow('agent_session_operation_unknown')
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
})
it('keeps the recorded handle when the restarted host no longer has the pane', async () => {
await launchUntilPasteStarts(hostRuntime())
await restartHost()
await expect(launch(hostRuntime(), PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toMatchObject({
outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY }
})
})
it('stays unknown when the host died before the surface was recorded', async () => {
const dying = hostRuntime()
let spawnRequested: () => void = () => {}
@@ -193,9 +297,12 @@ describe('a host restart mid-launch', () => {
expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
await restartHost()
const restarted = hostRuntime()
const restarted = restartedHostRuntime()
await expect(launch(restarted)).resolves.toEqual(first)
await expect(launch(restarted)).resolves.toEqual({
...first,
outcome: { ...first.outcome, handle: ADOPTED_HANDLE }
})
expect(restarted.createTerminal).not.toHaveBeenCalled()
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
})
@@ -221,6 +328,50 @@ describe('a host restart mid-launch', () => {
})
})
describe('a final write that fails', () => {
it('leaves the prompt unconfirmed, or unknown to a caller that cannot read it, never "not sent"', async () => {
replaceLaunchWrite(store, 2, () => Promise.reject(new Error('SQLITE_BUSY')))
const host = hostRuntime()
const first = await launch(host)
// The live answer is the truth the host saw; only the record missed it.
expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
await expect(launch(host)).rejects.toThrow('agent_session_operation_unknown')
await expect(launch(host, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toMatchObject({
outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY },
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
})
expect(host.createTerminal).toHaveBeenCalledOnce()
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
})
})
describe('the two writes', () => {
it('land in order when the paste returns before the first write has committed', async () => {
const recorded = vi.spyOn(store, 'recordOperationOutcome')
await launch(hostRuntime())
// Both were queued before either committed; the second is the one left standing.
const launchWrites = recorded.mock.calls.filter(([input]) => input.callerKey === 'device-1')
expect(launchWrites.map(([input]) => promptOutcome(input.outcome))).toEqual([
'unconfirmed',
'handed-to-terminal'
])
expect(promptOutcome(row()?.outcome)).toBe('handed-to-terminal')
})
it('never lets a failed first write block the launch', async () => {
replaceLaunchWrite(store, 1, () => Promise.reject(new Error('SQLITE_BUSY')))
const first = await launch(hostRuntime())
expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
expect(promptOutcome(row()?.outcome)).toBe('handed-to-terminal')
})
})
describe('a reply lost three times', () => {
it('starts one agent, and every retry gets its answer', async () => {
const host = hostRuntime()
@@ -276,6 +427,30 @@ describe('the desktop launches replay-safely', () => {
expect(row('trusted-local:desktop')?.outcome.status).toBe('succeeded')
})
it('reads an unconfirmed prompt after a restart mid-paste, through its own IPC transport', async () => {
const request = {
id: 'request-1',
authToken: 'desktop-ipc',
method: 'agent.launchReplay',
params: PROMPTED_LAUNCH
}
deliverTerminalPrompt.mockImplementationOnce(() => new Promise<boolean>(() => {}))
void dispatcherFor(hostRuntime()).dispatch(request, DESKTOP_IPC)
const deadline = Date.now() + 2_000
while (row('trusted-local:desktop')?.outcome.status !== 'succeeded') {
if (Date.now() > deadline) {
throw new Error('the desktop launch was never recorded')
}
await new Promise((resolve) => setTimeout(resolve, 5))
}
await ledgerWritesQueuedBefore(store)
await restartHost()
const replayed = await dispatcherFor(restartedHostRuntime()).dispatch(request, DESKTOP_IPC)
expect(replayed).toMatchObject({ ok: true, result: UNCONFIRMED_AGENT })
})
it('opens the ledger alone, never the chat host, to admit a terminal launch', async () => {
const host = hostRuntime()
@@ -39,6 +39,9 @@ export type AgentLaunchRuntimeStubOptions = {
terminalPaneAlreadyLive?: boolean
/** What the runtime reports about an offered prompt's typed line; unset reports nothing. */
lineCarriesPrompt?: boolean
/** Panes this runtime found already running, by the handle it issued them: a restarted host
* adopting a surviving PTY issues a new handle for the same pane. */
adoptedPanes?: Record<string, string>
}
function reportPromptCarry(
@@ -60,6 +63,8 @@ export function setAgentLaunchRecordStore(store: AgentSessionRecordStore | null)
export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
const worktreeCreateResults = new Map<string, Promise<unknown>>()
// Only the panes this runtime created or adopted: a handle is process-scoped, a pane key is not.
const handlesByPaneKey = new Map(Object.entries(options.adoptedPanes ?? {}))
const waitForSetupTerminalCompletion = vi.fn(
async (_handle: string, _signal?: AbortSignal): Promise<{ exitCode: number | null }> => ({
exitCode: 0
@@ -89,6 +94,9 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
showRepo: vi.fn(async () => ({ id: 'repo-1' })),
createManagedWorktree: vi.fn(async (args: Record<string, unknown>) => {
reportPromptCarry(options, args.onStartupPromptCarry, args.startupPrompt)
if (args.startupAgent && options.startupTerminalPaneKey) {
handlesByPaneKey.set(options.startupTerminalPaneKey, 'term_agent_first')
}
return {
worktree: { id: 'wt-new' },
startupTerminal: args.startupAgent
@@ -107,6 +115,9 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
throw new AgentLaunchPaneAlreadyLiveError()
}
reportPromptCarry(options, createOptions?.onStartupPromptCarry, createOptions?.startupPrompt)
if (options.terminalPaneKey) {
handlesByPaneKey.set(options.terminalPaneKey, 'term_1')
}
return {
handle: 'term_1',
...(options.terminalPaneKey ? { paneKey: options.terminalPaneKey } : {}),
@@ -114,6 +125,7 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
}
}),
showTerminal: vi.fn(async (handle: string) => ({ handle, worktreeId: 'wt-7' })),
getTerminalHandleForPaneKey: vi.fn((paneKey: string) => handlesByPaneKey.get(paneKey) ?? null),
isTerminalRunningAgent: vi.fn(async () => true),
showManagedTerminalWorkspace: vi.fn(async (selector: string) => ({
id: selector.replace(/^id:/, '')
+5 -3
View File
@@ -147,8 +147,9 @@ type ReplaySafeLaunch = {
attachOperationId: string
callerKey: string
terminalSpawn: TerminalSpawnDispatch
/** Records the surface the moment it exists. Fired, never awaited: the ledger's transactions run
* in order, so the final settle still lands after it, and the prompt never waits on bookkeeping. */
/** Records the surface the moment it exists, an owed prompt as `unconfirmed`. Fired, never
* awaited: the ledger's transactions run in order, so the final settle still lands after it, and
* the prompt never waits on bookkeeping. */
recordSurface: (provisional: AgentLaunchResult) => void
}
@@ -280,7 +281,8 @@ async function executeReplaySafeAgentLaunch(
}
throw new AgentLaunchExecutionError(error, failedWithoutEffects !== null)
}
// Settlement is bookkeeping; failure leaves the truthful `unknown` refusal for later retries.
// Bookkeeping: a failure leaves the first write, whose owed prompt replays as `unconfirmed` (or as
// `unknown` to a caller that cannot read it), never as `not-delivered`.
await settleQuietly(admission.settle(result))
return result
}
@@ -24,6 +24,11 @@ export const DESKTOP_RPC_CALLER: RpcCallerIdentity = { kind: 'desktop' }
* A transport that names its caller is believed; one that declares no client at all is the
* in-process runtime socket, trusted as it always was; one that declares a client it cannot name has
* no identity, and anything that needs one refuses it.
*
* Temporary: "no declared client" means the local CLI, so the SSH remote CLI bridge and browser
* automation share its namespace, and a new transport that forgets to stamp its caller inherits it.
* Before plugins ship (plan §4) the runtime socket stamps `local-cli` itself and no stamp means no
* identity.
*/
export function resolveRpcCallerIdentity(
transport:
+10 -1
View File
@@ -154,6 +154,13 @@ export type AgentLaunchPromptDisposal =
| { outcome: 'handed-to-terminal' }
/** Not delivered by this call; the caller still owns the text. */
| { outcome: 'not-delivered' }
/**
* Only ever replayed, never a live answer: the host recorded the running agent, then stopped
* before the delivery reported back, so the text may or may not have arrived. The caller must not
* resend. Sent only to a caller advertising `agent.launch.prompt-unconfirmed.v1`; every other
* caller is refused with `agent_session_operation_unknown` instead.
*/
| { outcome: 'unconfirmed' }
export type AgentLaunchPromptReceipt = {
delivery: AgentLaunchPromptDelivery
@@ -242,7 +249,9 @@ function isAgentLaunchPromptReceipt(value: unknown): value is AgentLaunchPromptR
}
return value.outcome === 'journaled'
? 'messageId' in value && typeof value.messageId === 'string'
: value.outcome === 'handed-to-terminal' || value.outcome === 'not-delivered'
: value.outcome === 'handed-to-terminal' ||
value.outcome === 'not-delivered' ||
value.outcome === 'unconfirmed'
}
function isAgentLaunchOutcome(value: unknown): value is AgentLaunchOutcome {
@@ -29,9 +29,15 @@ export const AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY = 'agent.launch.prompt
export const AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY =
'agent.launch.replay-required.v1' as const
// A client that reads `prompt.outcome: 'unconfirmed'` on a replayed launch. Without it, a launch
// whose host stopped mid-delivery replays as `agent_session_operation_unknown`, as it always has.
export const AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY =
'agent.launch.prompt-unconfirmed.v1' as const
export const AGENT_LAUNCH_RUNTIME_CAPABILITIES = [
AGENT_LAUNCH_RUNTIME_CAPABILITY,
AGENT_LAUNCH_REPLAY_RUNTIME_CAPABILITY,
AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY,
AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY
AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY,
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
] as const