mirror of
https://github.com/stablyai/orca.git
synced 2026-10-09 08:02:35 +00:00
fix(agent-launch): a replayed launch names its live terminal and never calls an unfinished prompt unsent
Two answers a retried launch got after a host restart were wrong. The recorded terminal handle belonged to the process that issued it, so after a restart it named nothing, even though the pane was still running. A replay now re-derives the handle from the recorded pane key in the current runtime. A pane that is gone keeps the recorded handle, which resolves to not-found; shipped clients require a handle, so it cannot be dropped. The record written when the tab appears said the prompt was "not delivered", and a failed final write left that standing. A host that stopped mid-paste cannot know that, and "not delivered" invites a resend that becomes a duplicate turn. The first write now records a prompt still owed as "unconfirmed". On replay, a caller that advertises agent.launch.prompt-unconfirmed.v1 (the desktop, and the CLI, which ships with the host) gets the running agent with that word; every other caller gets agent_session_operation_unknown, the answer it got before the first write existed. An older build reading the row rejects the word and answers the same.
This commit is contained in:
@@ -615,7 +615,7 @@ describe('the surface is published as the launch stands, before its prompt is de
|
||||
return { launch, published }
|
||||
}
|
||||
|
||||
it('records a prompt still owed as not delivered, then delivers it', async () => {
|
||||
it('records a prompt still owed as unconfirmed, then delivers it', async () => {
|
||||
const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: false })
|
||||
|
||||
const result = await launch.run(PROMPTED_EXISTING)
|
||||
@@ -626,12 +626,25 @@ describe('the surface is published as the launch stands, before its prompt is de
|
||||
outcome: { kind: 'terminal', handle: 'term_1' },
|
||||
worktreeId: 'wt-7',
|
||||
receipt: result.receipt,
|
||||
prompt: { delivery: 'submit', outcome: 'not-delivered' }
|
||||
// A host that stops mid-paste cannot say whether it landed, so it must not say "not sent".
|
||||
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
|
||||
}
|
||||
])
|
||||
expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
|
||||
})
|
||||
|
||||
it('records a draft as not delivered, since the host never delivers one', async () => {
|
||||
const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: false })
|
||||
|
||||
const result = await launch.run({
|
||||
...PROMPTED_EXISTING,
|
||||
prompt: { text: 'fix the build', delivery: 'draft' }
|
||||
})
|
||||
|
||||
expect(published[0]?.prompt).toEqual({ delivery: 'draft', outcome: 'not-delivered' })
|
||||
expect(result.prompt).toEqual({ delivery: 'draft', outcome: 'not-delivered' })
|
||||
})
|
||||
|
||||
it('records a prompt the launch command carried as already handed over', async () => {
|
||||
const { launch, published } = publishing({ settings: {}, lineCarriesPrompt: true })
|
||||
|
||||
@@ -655,7 +668,7 @@ describe('the surface is published as the launch stands, before its prompt is de
|
||||
])
|
||||
expect(published[0]).toEqual({
|
||||
...result,
|
||||
prompt: { delivery: 'submit', outcome: 'not-delivered' }
|
||||
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
|
||||
})
|
||||
expect(result.prompt).toEqual({ delivery: 'submit', outcome: 'journaled', messageId: 'msg-1' })
|
||||
})
|
||||
|
||||
@@ -77,8 +77,9 @@ export type AgentLaunchExecution = {
|
||||
|
||||
/**
|
||||
* The launch as it stands once its surface exists: a complete result whose prompt receipt says only
|
||||
* what creation itself settled — carried on the launch command, or not (yet) delivered. Complete so
|
||||
* a host that dies during the delivery still leaves a truthful answer behind.
|
||||
* what creation itself settled — carried on the launch command, a draft the host never delivers, or
|
||||
* a submit still `unconfirmed`. Complete so a host that dies during the delivery still leaves a
|
||||
* truthful answer behind.
|
||||
*/
|
||||
export type AgentLaunchPublishedSurface = AgentLaunchResult
|
||||
|
||||
@@ -109,7 +110,7 @@ export async function executeAgentLaunch(
|
||||
outcome: { kind: 'terminal', handle: intent.reuseTerminal.handle },
|
||||
worktreeId: existingWorktreeId(intent.target),
|
||||
receipt: preflight,
|
||||
...promptReceipt(intent, settledAtCreation({}))
|
||||
...promptReceipt(intent, settledAtCreation(intent, {}))
|
||||
})
|
||||
return {
|
||||
...reused,
|
||||
@@ -134,7 +135,7 @@ export async function executeAgentLaunch(
|
||||
worktreeId: placed.worktreeId,
|
||||
receipt: preflight,
|
||||
...(placed.warning ? { warning: placed.warning } : {}),
|
||||
...promptReceipt(intent, settledAtCreation(placed))
|
||||
...promptReceipt(intent, settledAtCreation(intent, placed))
|
||||
})
|
||||
return {
|
||||
...startup,
|
||||
@@ -192,7 +193,7 @@ export async function executeAgentLaunch(
|
||||
worktreeId: placed.worktreeId,
|
||||
receipt: settled,
|
||||
...(warning ? { warning } : {}),
|
||||
...promptReceipt(intent, settledAtCreation(created))
|
||||
...promptReceipt(intent, settledAtCreation(intent, created))
|
||||
})
|
||||
return {
|
||||
...surface,
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
* terminal, line fits -> folded into the command that execs the agent -> handed-to-terminal
|
||||
* terminal, otherwise -> bracketed paste into the live PTY once it is ready -> handed-to-terminal
|
||||
* anything unproven -> -> not-delivered
|
||||
* host stopped mid-delivery (a replayed record only) -> unconfirmed
|
||||
*
|
||||
* argv has no readiness race, so it is offered wherever the agent's CLI takes a prompt argument
|
||||
* (`agentPromptRidesLaunchCommand`). But that command is TYPED into the user's shell, and a long or
|
||||
@@ -28,16 +29,22 @@ import type { AgentLaunchStructuredSurface } from './agent-launch-surface-factor
|
||||
|
||||
export const HANDED_TO_TERMINAL: AgentLaunchPromptDisposal = { outcome: 'handed-to-terminal' }
|
||||
const NOT_DELIVERED: AgentLaunchPromptDisposal = { outcome: 'not-delivered' }
|
||||
const UNCONFIRMED: AgentLaunchPromptDisposal = { outcome: 'unconfirmed' }
|
||||
|
||||
/**
|
||||
* What the receipt may say before any delivery runs: only what creating the surface already
|
||||
* settled. A launch command that carried the text has handed it over; anything else is not (yet)
|
||||
* delivered, which is also the truthful answer when the host dies before delivering it.
|
||||
* What the record may say before any delivery runs: only what creating the surface already settled.
|
||||
* A launch command that carried the text has handed it over, and a draft is never delivered by the
|
||||
* host. A submit still owed is `unconfirmed`: a host that stops mid-delivery cannot say whether the
|
||||
* paste or the commit landed, and "not delivered" would invite a duplicate turn.
|
||||
*/
|
||||
export function settledAtCreation(created: {
|
||||
promptRodeLaunchCommand?: boolean
|
||||
}): AgentLaunchPromptDisposal {
|
||||
return created.promptRodeLaunchCommand ? HANDED_TO_TERMINAL : NOT_DELIVERED
|
||||
export function settledAtCreation(
|
||||
intent: Pick<AgentLaunchIntent, 'prompt'>,
|
||||
created: { promptRodeLaunchCommand?: boolean }
|
||||
): AgentLaunchPromptDisposal {
|
||||
if (created.promptRodeLaunchCommand) {
|
||||
return HANDED_TO_TERMINAL
|
||||
}
|
||||
return intent.prompt?.delivery === 'submit' ? UNCONFIRMED : NOT_DELIVERED
|
||||
}
|
||||
|
||||
/** Each surface delivers its own way, so the disposal is decided where the surface is known. */
|
||||
@@ -138,9 +145,9 @@ export function launchCommandPrompt(
|
||||
* Each arm is a consequence of the act it names, never a write-ahead of it: `journaled` is
|
||||
* reachable only from a committed message id, `handed-to-terminal` only from a launch command that
|
||||
* carried the text or a PTY write that returned, and everything else under-claims as
|
||||
* `not-delivered`. There is deliberately no arm for "maybe" — a caller holding one could neither
|
||||
* resend nor drop the text. Dispatch doubt is not this tier's to report: the submission row
|
||||
* carries it.
|
||||
* `not-delivered`. A live answer has no "maybe": `unconfirmed` is written only into the record
|
||||
* before delivery runs (`settledAtCreation`), and is read back only by a replay. Dispatch doubt is
|
||||
* not this tier's to report: the submission row carries it.
|
||||
*/
|
||||
export function promptReceipt(
|
||||
intent: AgentLaunchIntent,
|
||||
|
||||
@@ -26,7 +26,10 @@ import {
|
||||
WORKTREE_VISIBILITY_SOURCE_DEFAULTS_RUNTIME_CAPABILITY,
|
||||
type RuntimeCapability
|
||||
} from '../../shared/protocol-version'
|
||||
import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability'
|
||||
import {
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY
|
||||
} from '../../shared/agent-launch-runtime-capability'
|
||||
import { ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES } from '../../shared/electron-remote-runtime-client-capabilities'
|
||||
import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability'
|
||||
import { supportsAgentLaunch } from '../runtime/rpc/methods/agent-launch'
|
||||
@@ -58,7 +61,7 @@ const REMOTE_ONLY_BY_DECISION: readonly RuntimeCapability[] = [
|
||||
]
|
||||
|
||||
/** Gates the renderer must pass against its own main process. The Electron remote list omits all
|
||||
* seven; mobile advertises the structured ones, so this is an Electron-remote gap rather than a
|
||||
* eight; mobile advertises the structured ones, so this is an Electron-remote gap rather than a
|
||||
* statement that no remote client wants them. Why it is one is not recorded here. */
|
||||
const LOCAL_ONLY_BY_DECISION: readonly RuntimeCapability[] = [
|
||||
AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY,
|
||||
@@ -68,7 +71,9 @@ const LOCAL_ONLY_BY_DECISION: readonly RuntimeCapability[] = [
|
||||
STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY,
|
||||
CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY,
|
||||
// The desktop picks each launch mode itself; only its own host is told so far.
|
||||
STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY
|
||||
STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY,
|
||||
// Read by the desktop's own launches first; a remote host is told when its launches move over.
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
|
||||
]
|
||||
|
||||
function missingFrom(
|
||||
|
||||
@@ -9,7 +9,10 @@ import {
|
||||
STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY,
|
||||
type RuntimeCapability
|
||||
} from '../../shared/protocol-version'
|
||||
import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../shared/agent-launch-runtime-capability'
|
||||
import {
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY
|
||||
} from '../../shared/agent-launch-runtime-capability'
|
||||
import { AGENT_SESSION_BACKGROUND_TASK_CHILD_VIEWS_CAPABILITY } from '../../shared/agent-session-background-task-child-views-capability'
|
||||
|
||||
/**
|
||||
@@ -37,5 +40,8 @@ export const DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES: readonly RuntimeCapab
|
||||
STRUCTURED_AGENT_SESSION_CLIENT_LAUNCH_MODE_CAPABILITY,
|
||||
// Without this `supportsAgentLaunch` refuses the renderer outright, while the same renderer
|
||||
// targeting a remote host is admitted — the asymmetry this constant exists to close.
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY,
|
||||
// A replay after a restart mid-delivery answers the running agent with an `unconfirmed` prompt,
|
||||
// which the desktop reads, rather than refusing it as unknown.
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
|
||||
] as const
|
||||
|
||||
@@ -5,6 +5,11 @@
|
||||
// whole chat host (provider adapters, the host, the model catalog) first. The store is opened here
|
||||
// instead and the chat host, when something needs it, is built on this same instance: the store is
|
||||
// a single writer, so a second copy of it would diverge.
|
||||
//
|
||||
// Known edges, both reachable only once quit has begun (stop has no other production caller): a
|
||||
// stop whose host teardown fails empties the slot while that host still holds its store open, so a
|
||||
// later admission would open a second one; and a store opened after stop is closed by nothing but
|
||||
// process exit.
|
||||
|
||||
import { AgentSessionRecordStore } from './agent-session-record-store'
|
||||
import type { JournalHostDatabase } from '../native-chat/agent-session-journal/journal-host-database'
|
||||
|
||||
@@ -151,8 +151,9 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStartTuiIdleVis
|
||||
/**
|
||||
* Installs the structured agent-session host on first use. Lazy for the same
|
||||
* reason the orchestration DB is: the profile's user-data path is not final
|
||||
* until the app is ready, and a runtime nobody drives a chat session on
|
||||
* should never open the record store.
|
||||
* until the app is ready, and a runtime nobody drives a chat session on never
|
||||
* builds the chat host. The record store it sits on may already be open, from
|
||||
* a launch's admission.
|
||||
*/
|
||||
async ensureStructuredAgentSessionHost(): Promise<void> {
|
||||
await installStructuredAgentSessionHost({
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
*/
|
||||
|
||||
import { deriveAgentLaunchChildOperationId } from '../../../../shared/agent-launch-operation'
|
||||
import { AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY } from '../../../../shared/agent-launch-runtime-capability'
|
||||
import { isAgentLaunchResult, type AgentLaunchResult } from '../../../../shared/agent-launch-intent'
|
||||
import type {
|
||||
AgentSessionOperationOutcome,
|
||||
@@ -100,6 +101,60 @@ function answerFromRecordedRow(
|
||||
: { decision: 'refuse', refusal: replay.refusal }
|
||||
}
|
||||
|
||||
/** The CLI (no declared client) ships with this host; any other caller must say it reads the word. */
|
||||
function readsUnconfirmedLaunchPrompt(
|
||||
context: Pick<RpcContext, 'clientKind' | 'clientCapabilities'>
|
||||
): boolean {
|
||||
return (
|
||||
context.clientKind === undefined ||
|
||||
context.clientCapabilities?.includes(AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY) ===
|
||||
true
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* A terminal handle is issued by the process that answered, so a recorded one is dead after a
|
||||
* restart. The pane key is the durable name: the handle is re-derived from it in this runtime. A
|
||||
* pane this runtime no longer knows keeps the recorded handle, which then resolves to not-found —
|
||||
* the truth about a terminal that is gone. The handle stays because shipped clients require one.
|
||||
*/
|
||||
function withLiveTerminalHandle(
|
||||
recorded: AgentLaunchResult,
|
||||
runtime: Pick<RpcContext['runtime'], 'getTerminalHandleForPaneKey'>
|
||||
): AgentLaunchResult {
|
||||
const { outcome } = recorded
|
||||
if (outcome.kind !== 'terminal' || !outcome.paneKey) {
|
||||
return recorded
|
||||
}
|
||||
const handle = runtime.getTerminalHandleForPaneKey(outcome.paneKey)
|
||||
return handle && handle !== outcome.handle
|
||||
? { ...recorded, outcome: { ...outcome, handle } }
|
||||
: recorded
|
||||
}
|
||||
|
||||
/**
|
||||
* A recorded answer as this caller may read it now. A prompt the host stopped delivering is
|
||||
* `unconfirmed` only to a caller that reads the word; every other caller gets the refusal it got
|
||||
* before the first write existed, never a `not-delivered` that invites a duplicate turn.
|
||||
*/
|
||||
function presentRecordedAnswer(
|
||||
context: RpcContext,
|
||||
operationId: string,
|
||||
answer: AgentLaunchAdmission
|
||||
): AgentLaunchAdmission {
|
||||
if (answer.decision !== 'replay') {
|
||||
return answer
|
||||
}
|
||||
if (answer.result.prompt?.outcome === 'unconfirmed' && !readsUnconfirmedLaunchPrompt(context)) {
|
||||
return refusal(
|
||||
operationId,
|
||||
'agent_session_operation_unknown',
|
||||
'started its agent, but whether its prompt arrived is unknown'
|
||||
)
|
||||
}
|
||||
return { decision: 'replay', result: withLiveTerminalHandle(answer.result, context.runtime) }
|
||||
}
|
||||
|
||||
/**
|
||||
* Admit, then claim, in one durable transaction so a launch writes the ledger once before its effect.
|
||||
*
|
||||
@@ -136,7 +191,7 @@ export async function admitAgentLaunchOperation(
|
||||
if (admitted.decision === 'replay') {
|
||||
const answer = answerFromRecordedRow(operationId, admitted.row.outcome)
|
||||
if (answer) {
|
||||
return answer
|
||||
return presentRecordedAnswer(context, operationId, answer)
|
||||
}
|
||||
}
|
||||
// Unreachable with both steps in one transaction; answered as uncertain rather than run twice.
|
||||
@@ -150,10 +205,10 @@ export async function admitAgentLaunchOperation(
|
||||
if (claim.claim === 'lost') {
|
||||
// The handler joins same-process retries before admission. Reaching a claimed row here means
|
||||
// this runtime did not start it, so treating it as restart uncertainty is the safe answer.
|
||||
return (
|
||||
answerFromRecordedRow(operationId, claim.row.outcome) ??
|
||||
refusal(operationId, 'agent_session_operation_unknown', 'is claimed but unsettled')
|
||||
)
|
||||
const answer = answerFromRecordedRow(operationId, claim.row.outcome)
|
||||
return answer
|
||||
? presentRecordedAnswer(context, operationId, answer)
|
||||
: refusal(operationId, 'agent_session_operation_unknown', 'is claimed but unsettled')
|
||||
}
|
||||
const succeeded = (result: AgentLaunchResult) =>
|
||||
store.recordOperationOutcome({
|
||||
|
||||
@@ -4,15 +4,19 @@
|
||||
* The record is written twice: once when the surface exists and once when the prompt's fate is
|
||||
* known. A host that dies at any point leaves the replay a truthful answer — the running agent once
|
||||
* its surface is recorded, an honest "unknown" before — and never a second agent. A "restart" here
|
||||
* is what a new process sees: the store reopened from disk and a runtime with no in-flight launches.
|
||||
* is what a new process sees: the store reopened from disk and a runtime with no in-flight launches,
|
||||
* which issues its own handles, so a surviving pane comes back under a new one.
|
||||
*/
|
||||
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { AGENT_LAUNCH_RUNTIME_CAPABILITY } from '../../../../shared/agent-launch-runtime-capability'
|
||||
import type { AgentLaunchResult } from '../../../../shared/agent-launch-intent'
|
||||
import {
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY
|
||||
} from '../../../../shared/agent-launch-runtime-capability'
|
||||
import { isAgentLaunchResult, type AgentLaunchResult } from '../../../../shared/agent-launch-intent'
|
||||
import type { AgentSessionOperationRow } from '../../../../shared/agent-session-operation-ledger'
|
||||
import type { AgentSessionRecordStore } from '../../agent-session-record-store'
|
||||
import { openTestAgentSessionRecordStore } from '../../agent-session-record-store-test-harness'
|
||||
@@ -20,6 +24,7 @@ import type { OrcaRuntimeService } from '../../orca-runtime'
|
||||
import type { RpcContext } from '../core'
|
||||
import { RpcDispatcher } from '../dispatcher'
|
||||
import { DESKTOP_RPC_CALLER } from '../rpc-caller-identity'
|
||||
import { DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES } from '../../../ipc/desktop-renderer-runtime-capabilities'
|
||||
import {
|
||||
methodNamed,
|
||||
rpcContext,
|
||||
@@ -54,22 +59,45 @@ const PHONE: Partial<RpcContext> = {
|
||||
pairedDeviceId: 'device-1',
|
||||
clientCapabilities: [AGENT_LAUNCH_RUNTIME_CAPABILITY]
|
||||
}
|
||||
/** The same paired device on a build that reads an `unconfirmed` prompt. */
|
||||
const UPGRADED_PHONE: Partial<RpcContext> = {
|
||||
...PHONE,
|
||||
clientCapabilities: [
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
|
||||
]
|
||||
}
|
||||
/** Exactly what the desktop's `runtime:call` handler hands the dispatcher. */
|
||||
const DESKTOP_IPC = {
|
||||
clientId: 'desktop-renderer',
|
||||
caller: DESKTOP_RPC_CALLER,
|
||||
clientKind: 'runtime' as const,
|
||||
clientCapabilities: [AGENT_LAUNCH_RUNTIME_CAPABILITY]
|
||||
clientCapabilities: DESKTOP_RENDERER_RUNTIME_CLIENT_CAPABILITIES
|
||||
}
|
||||
/** The handle a restarted host issues the pane that outlived the old one. */
|
||||
const ADOPTED_HANDLE = 'term_adopted'
|
||||
|
||||
let directory: string
|
||||
let store: AgentSessionRecordStore
|
||||
|
||||
function hostRuntime() {
|
||||
function hostRuntime(adoptedPanes?: Record<string, string>) {
|
||||
// A phone's launch into an existing workspace also moves the phone's own view to the new tab.
|
||||
return Object.assign(runtimeStub({ settings: TERMINAL_ONLY, terminalPaneKey: PANE_KEY }), {
|
||||
selectCreatedMobileSessionTabForClient: vi.fn(() => true)
|
||||
})
|
||||
return Object.assign(
|
||||
runtimeStub({ settings: TERMINAL_ONLY, terminalPaneKey: PANE_KEY, adoptedPanes }),
|
||||
{ selectCreatedMobileSessionTabForClient: vi.fn(() => true) }
|
||||
)
|
||||
}
|
||||
|
||||
/** The host after a restart, still running the pane the dead one launched. */
|
||||
function restartedHostRuntime() {
|
||||
return hostRuntime({ [PANE_KEY]: ADOPTED_HANDLE })
|
||||
}
|
||||
|
||||
const UNCONFIRMED_AGENT = {
|
||||
outcome: { kind: 'terminal', handle: ADOPTED_HANDLE, paneKey: PANE_KEY },
|
||||
worktreeId: 'wt-7',
|
||||
receipt: expect.objectContaining({ mode: 'terminal' }),
|
||||
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
|
||||
}
|
||||
|
||||
function launch(
|
||||
@@ -113,6 +141,46 @@ async function ledgerWritesQueuedBefore(ledger: AgentSessionRecordStore): Promis
|
||||
})
|
||||
}
|
||||
|
||||
/** Launches with a paste that never returns: the host dies while it waits for the agent. */
|
||||
async function launchUntilPasteStarts(runtime: AgentLaunchRuntimeStub): Promise<void> {
|
||||
let pasting: () => void = () => {}
|
||||
const pasteStarted = new Promise<void>((resolve) => {
|
||||
pasting = resolve
|
||||
})
|
||||
deliverTerminalPrompt.mockImplementationOnce(() => {
|
||||
pasting()
|
||||
return new Promise<boolean>(() => {})
|
||||
})
|
||||
void launch(runtime)
|
||||
await pasteStarted
|
||||
await ledgerWritesQueuedBefore(store)
|
||||
}
|
||||
|
||||
/** The ledger as the launch reaches it, with the launch's `nth` write (1-based) replaced. */
|
||||
function replaceLaunchWrite(
|
||||
ledger: AgentSessionRecordStore,
|
||||
nth: number,
|
||||
write: () => Promise<void>
|
||||
): void {
|
||||
const recordOperationOutcome = ledger.recordOperationOutcome.bind(ledger)
|
||||
let launchWrites = 0
|
||||
vi.spyOn(ledger, 'recordOperationOutcome').mockImplementation((input) => {
|
||||
if (input.operationId === OPERATION_ID && input.callerKey === 'device-1') {
|
||||
launchWrites += 1
|
||||
if (launchWrites === nth) {
|
||||
return write()
|
||||
}
|
||||
}
|
||||
return recordOperationOutcome(input)
|
||||
})
|
||||
}
|
||||
|
||||
function promptOutcome(outcome: AgentSessionOperationRow['outcome'] | undefined): unknown {
|
||||
return outcome?.status === 'succeeded' && isAgentLaunchResult(outcome.launch)
|
||||
? outcome.launch.prompt?.outcome
|
||||
: undefined
|
||||
}
|
||||
|
||||
function dispatcherFor(runtime: AgentLaunchRuntimeStub): RpcDispatcher {
|
||||
return new RpcDispatcher({
|
||||
// oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: the fixture implements every runtime method agent.launch reaches, plus the id the dispatcher stamps on replies.
|
||||
@@ -135,37 +203,73 @@ afterEach(async () => {
|
||||
})
|
||||
|
||||
describe('a host restart mid-launch', () => {
|
||||
it('finds the running agent when the host died while its prompt waited for readiness', async () => {
|
||||
// The paste waits for the agent forever: the host dies first.
|
||||
let waiting: () => void = () => {}
|
||||
const readinessWait = new Promise<void>((resolve) => {
|
||||
waiting = resolve
|
||||
})
|
||||
deliverTerminalPrompt.mockImplementationOnce(() => {
|
||||
waiting()
|
||||
return new Promise<boolean>(() => {})
|
||||
})
|
||||
it('finds the running agent, under the handle the new host issued, after a restart mid-paste', async () => {
|
||||
const dying = hostRuntime()
|
||||
void launch(dying)
|
||||
await readinessWait
|
||||
await ledgerWritesQueuedBefore(store)
|
||||
await launchUntilPasteStarts(dying)
|
||||
|
||||
await restartHost()
|
||||
const restarted = hostRuntime()
|
||||
const replayed = await launch(restarted)
|
||||
const restarted = restartedHostRuntime()
|
||||
|
||||
expect(replayed).toEqual({
|
||||
outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY },
|
||||
worktreeId: 'wt-7',
|
||||
receipt: expect.objectContaining({ mode: 'terminal' }),
|
||||
// The dead host never pasted it, and saying so lets the caller keep the text.
|
||||
prompt: { delivery: 'submit', outcome: 'not-delivered' }
|
||||
})
|
||||
// The dead host's `term_1` names nothing now; the pane is the durable name.
|
||||
await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual(
|
||||
UNCONFIRMED_AGENT
|
||||
)
|
||||
expect(restarted.createTerminal).not.toHaveBeenCalled()
|
||||
expect(dying.createTerminal).toHaveBeenCalledOnce()
|
||||
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
|
||||
})
|
||||
|
||||
it('answers a caller that cannot read "unconfirmed" as before: unknown, never "not sent"', async () => {
|
||||
await launchUntilPasteStarts(hostRuntime())
|
||||
|
||||
await restartHost()
|
||||
const restarted = restartedHostRuntime()
|
||||
|
||||
await expect(launch(restarted)).rejects.toThrow('agent_session_operation_unknown')
|
||||
// The same row still answers a caller that reads the word: the refusal is presentation only.
|
||||
await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual(
|
||||
UNCONFIRMED_AGENT
|
||||
)
|
||||
expect(restarted.createTerminal).not.toHaveBeenCalled()
|
||||
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
|
||||
})
|
||||
|
||||
it('stays unconfirmed when the host died after the paste returned, before the final write', async () => {
|
||||
let pasted: () => void = () => {}
|
||||
const pasteReturned = new Promise<void>((resolve) => {
|
||||
pasted = resolve
|
||||
})
|
||||
deliverTerminalPrompt.mockImplementationOnce(async () => {
|
||||
pasted()
|
||||
return true
|
||||
})
|
||||
// The final write never lands: the process is gone by then.
|
||||
replaceLaunchWrite(store, 2, () => new Promise<void>(() => {}))
|
||||
void launch(hostRuntime())
|
||||
await pasteReturned
|
||||
await untilRecorded('succeeded')
|
||||
await ledgerWritesQueuedBefore(store)
|
||||
|
||||
await restartHost()
|
||||
const restarted = restartedHostRuntime()
|
||||
|
||||
await expect(launch(restarted, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toEqual(
|
||||
UNCONFIRMED_AGENT
|
||||
)
|
||||
await expect(launch(restarted)).rejects.toThrow('agent_session_operation_unknown')
|
||||
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
|
||||
})
|
||||
|
||||
it('keeps the recorded handle when the restarted host no longer has the pane', async () => {
|
||||
await launchUntilPasteStarts(hostRuntime())
|
||||
|
||||
await restartHost()
|
||||
|
||||
await expect(launch(hostRuntime(), PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toMatchObject({
|
||||
outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY }
|
||||
})
|
||||
})
|
||||
|
||||
it('stays unknown when the host died before the surface was recorded', async () => {
|
||||
const dying = hostRuntime()
|
||||
let spawnRequested: () => void = () => {}
|
||||
@@ -193,9 +297,12 @@ describe('a host restart mid-launch', () => {
|
||||
expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
|
||||
|
||||
await restartHost()
|
||||
const restarted = hostRuntime()
|
||||
const restarted = restartedHostRuntime()
|
||||
|
||||
await expect(launch(restarted)).resolves.toEqual(first)
|
||||
await expect(launch(restarted)).resolves.toEqual({
|
||||
...first,
|
||||
outcome: { ...first.outcome, handle: ADOPTED_HANDLE }
|
||||
})
|
||||
expect(restarted.createTerminal).not.toHaveBeenCalled()
|
||||
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
|
||||
})
|
||||
@@ -221,6 +328,50 @@ describe('a host restart mid-launch', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('a final write that fails', () => {
|
||||
it('leaves the prompt unconfirmed, or unknown to a caller that cannot read it, never "not sent"', async () => {
|
||||
replaceLaunchWrite(store, 2, () => Promise.reject(new Error('SQLITE_BUSY')))
|
||||
const host = hostRuntime()
|
||||
|
||||
const first = await launch(host)
|
||||
|
||||
// The live answer is the truth the host saw; only the record missed it.
|
||||
expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
|
||||
await expect(launch(host)).rejects.toThrow('agent_session_operation_unknown')
|
||||
await expect(launch(host, PROMPTED_LAUNCH, UPGRADED_PHONE)).resolves.toMatchObject({
|
||||
outcome: { kind: 'terminal', handle: 'term_1', paneKey: PANE_KEY },
|
||||
prompt: { delivery: 'submit', outcome: 'unconfirmed' }
|
||||
})
|
||||
expect(host.createTerminal).toHaveBeenCalledOnce()
|
||||
expect(deliverTerminalPrompt).toHaveBeenCalledOnce()
|
||||
})
|
||||
})
|
||||
|
||||
describe('the two writes', () => {
|
||||
it('land in order when the paste returns before the first write has committed', async () => {
|
||||
const recorded = vi.spyOn(store, 'recordOperationOutcome')
|
||||
|
||||
await launch(hostRuntime())
|
||||
|
||||
// Both were queued before either committed; the second is the one left standing.
|
||||
const launchWrites = recorded.mock.calls.filter(([input]) => input.callerKey === 'device-1')
|
||||
expect(launchWrites.map(([input]) => promptOutcome(input.outcome))).toEqual([
|
||||
'unconfirmed',
|
||||
'handed-to-terminal'
|
||||
])
|
||||
expect(promptOutcome(row()?.outcome)).toBe('handed-to-terminal')
|
||||
})
|
||||
|
||||
it('never lets a failed first write block the launch', async () => {
|
||||
replaceLaunchWrite(store, 1, () => Promise.reject(new Error('SQLITE_BUSY')))
|
||||
|
||||
const first = await launch(hostRuntime())
|
||||
|
||||
expect(first.prompt).toEqual({ delivery: 'submit', outcome: 'handed-to-terminal' })
|
||||
expect(promptOutcome(row()?.outcome)).toBe('handed-to-terminal')
|
||||
})
|
||||
})
|
||||
|
||||
describe('a reply lost three times', () => {
|
||||
it('starts one agent, and every retry gets its answer', async () => {
|
||||
const host = hostRuntime()
|
||||
@@ -276,6 +427,30 @@ describe('the desktop launches replay-safely', () => {
|
||||
expect(row('trusted-local:desktop')?.outcome.status).toBe('succeeded')
|
||||
})
|
||||
|
||||
it('reads an unconfirmed prompt after a restart mid-paste, through its own IPC transport', async () => {
|
||||
const request = {
|
||||
id: 'request-1',
|
||||
authToken: 'desktop-ipc',
|
||||
method: 'agent.launchReplay',
|
||||
params: PROMPTED_LAUNCH
|
||||
}
|
||||
deliverTerminalPrompt.mockImplementationOnce(() => new Promise<boolean>(() => {}))
|
||||
void dispatcherFor(hostRuntime()).dispatch(request, DESKTOP_IPC)
|
||||
const deadline = Date.now() + 2_000
|
||||
while (row('trusted-local:desktop')?.outcome.status !== 'succeeded') {
|
||||
if (Date.now() > deadline) {
|
||||
throw new Error('the desktop launch was never recorded')
|
||||
}
|
||||
await new Promise((resolve) => setTimeout(resolve, 5))
|
||||
}
|
||||
await ledgerWritesQueuedBefore(store)
|
||||
|
||||
await restartHost()
|
||||
const replayed = await dispatcherFor(restartedHostRuntime()).dispatch(request, DESKTOP_IPC)
|
||||
|
||||
expect(replayed).toMatchObject({ ok: true, result: UNCONFIRMED_AGENT })
|
||||
})
|
||||
|
||||
it('opens the ledger alone, never the chat host, to admit a terminal launch', async () => {
|
||||
const host = hostRuntime()
|
||||
|
||||
|
||||
@@ -39,6 +39,9 @@ export type AgentLaunchRuntimeStubOptions = {
|
||||
terminalPaneAlreadyLive?: boolean
|
||||
/** What the runtime reports about an offered prompt's typed line; unset reports nothing. */
|
||||
lineCarriesPrompt?: boolean
|
||||
/** Panes this runtime found already running, by the handle it issued them: a restarted host
|
||||
* adopting a surviving PTY issues a new handle for the same pane. */
|
||||
adoptedPanes?: Record<string, string>
|
||||
}
|
||||
|
||||
function reportPromptCarry(
|
||||
@@ -60,6 +63,8 @@ export function setAgentLaunchRecordStore(store: AgentSessionRecordStore | null)
|
||||
|
||||
export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
|
||||
const worktreeCreateResults = new Map<string, Promise<unknown>>()
|
||||
// Only the panes this runtime created or adopted: a handle is process-scoped, a pane key is not.
|
||||
const handlesByPaneKey = new Map(Object.entries(options.adoptedPanes ?? {}))
|
||||
const waitForSetupTerminalCompletion = vi.fn(
|
||||
async (_handle: string, _signal?: AbortSignal): Promise<{ exitCode: number | null }> => ({
|
||||
exitCode: 0
|
||||
@@ -89,6 +94,9 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
|
||||
showRepo: vi.fn(async () => ({ id: 'repo-1' })),
|
||||
createManagedWorktree: vi.fn(async (args: Record<string, unknown>) => {
|
||||
reportPromptCarry(options, args.onStartupPromptCarry, args.startupPrompt)
|
||||
if (args.startupAgent && options.startupTerminalPaneKey) {
|
||||
handlesByPaneKey.set(options.startupTerminalPaneKey, 'term_agent_first')
|
||||
}
|
||||
return {
|
||||
worktree: { id: 'wt-new' },
|
||||
startupTerminal: args.startupAgent
|
||||
@@ -107,6 +115,9 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
|
||||
throw new AgentLaunchPaneAlreadyLiveError()
|
||||
}
|
||||
reportPromptCarry(options, createOptions?.onStartupPromptCarry, createOptions?.startupPrompt)
|
||||
if (options.terminalPaneKey) {
|
||||
handlesByPaneKey.set(options.terminalPaneKey, 'term_1')
|
||||
}
|
||||
return {
|
||||
handle: 'term_1',
|
||||
...(options.terminalPaneKey ? { paneKey: options.terminalPaneKey } : {}),
|
||||
@@ -114,6 +125,7 @@ export function runtimeStub(options: AgentLaunchRuntimeStubOptions = {}) {
|
||||
}
|
||||
}),
|
||||
showTerminal: vi.fn(async (handle: string) => ({ handle, worktreeId: 'wt-7' })),
|
||||
getTerminalHandleForPaneKey: vi.fn((paneKey: string) => handlesByPaneKey.get(paneKey) ?? null),
|
||||
isTerminalRunningAgent: vi.fn(async () => true),
|
||||
showManagedTerminalWorkspace: vi.fn(async (selector: string) => ({
|
||||
id: selector.replace(/^id:/, '')
|
||||
|
||||
@@ -147,8 +147,9 @@ type ReplaySafeLaunch = {
|
||||
attachOperationId: string
|
||||
callerKey: string
|
||||
terminalSpawn: TerminalSpawnDispatch
|
||||
/** Records the surface the moment it exists. Fired, never awaited: the ledger's transactions run
|
||||
* in order, so the final settle still lands after it, and the prompt never waits on bookkeeping. */
|
||||
/** Records the surface the moment it exists, an owed prompt as `unconfirmed`. Fired, never
|
||||
* awaited: the ledger's transactions run in order, so the final settle still lands after it, and
|
||||
* the prompt never waits on bookkeeping. */
|
||||
recordSurface: (provisional: AgentLaunchResult) => void
|
||||
}
|
||||
|
||||
@@ -280,7 +281,8 @@ async function executeReplaySafeAgentLaunch(
|
||||
}
|
||||
throw new AgentLaunchExecutionError(error, failedWithoutEffects !== null)
|
||||
}
|
||||
// Settlement is bookkeeping; failure leaves the truthful `unknown` refusal for later retries.
|
||||
// Bookkeeping: a failure leaves the first write, whose owed prompt replays as `unconfirmed` (or as
|
||||
// `unknown` to a caller that cannot read it), never as `not-delivered`.
|
||||
await settleQuietly(admission.settle(result))
|
||||
return result
|
||||
}
|
||||
|
||||
@@ -24,6 +24,11 @@ export const DESKTOP_RPC_CALLER: RpcCallerIdentity = { kind: 'desktop' }
|
||||
* A transport that names its caller is believed; one that declares no client at all is the
|
||||
* in-process runtime socket, trusted as it always was; one that declares a client it cannot name has
|
||||
* no identity, and anything that needs one refuses it.
|
||||
*
|
||||
* Temporary: "no declared client" means the local CLI, so the SSH remote CLI bridge and browser
|
||||
* automation share its namespace, and a new transport that forgets to stamp its caller inherits it.
|
||||
* Before plugins ship (plan §4) the runtime socket stamps `local-cli` itself and no stamp means no
|
||||
* identity.
|
||||
*/
|
||||
export function resolveRpcCallerIdentity(
|
||||
transport:
|
||||
|
||||
@@ -154,6 +154,13 @@ export type AgentLaunchPromptDisposal =
|
||||
| { outcome: 'handed-to-terminal' }
|
||||
/** Not delivered by this call; the caller still owns the text. */
|
||||
| { outcome: 'not-delivered' }
|
||||
/**
|
||||
* Only ever replayed, never a live answer: the host recorded the running agent, then stopped
|
||||
* before the delivery reported back, so the text may or may not have arrived. The caller must not
|
||||
* resend. Sent only to a caller advertising `agent.launch.prompt-unconfirmed.v1`; every other
|
||||
* caller is refused with `agent_session_operation_unknown` instead.
|
||||
*/
|
||||
| { outcome: 'unconfirmed' }
|
||||
|
||||
export type AgentLaunchPromptReceipt = {
|
||||
delivery: AgentLaunchPromptDelivery
|
||||
@@ -242,7 +249,9 @@ function isAgentLaunchPromptReceipt(value: unknown): value is AgentLaunchPromptR
|
||||
}
|
||||
return value.outcome === 'journaled'
|
||||
? 'messageId' in value && typeof value.messageId === 'string'
|
||||
: value.outcome === 'handed-to-terminal' || value.outcome === 'not-delivered'
|
||||
: value.outcome === 'handed-to-terminal' ||
|
||||
value.outcome === 'not-delivered' ||
|
||||
value.outcome === 'unconfirmed'
|
||||
}
|
||||
|
||||
function isAgentLaunchOutcome(value: unknown): value is AgentLaunchOutcome {
|
||||
|
||||
@@ -29,9 +29,15 @@ export const AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY = 'agent.launch.prompt
|
||||
export const AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY =
|
||||
'agent.launch.replay-required.v1' as const
|
||||
|
||||
// A client that reads `prompt.outcome: 'unconfirmed'` on a replayed launch. Without it, a launch
|
||||
// whose host stopped mid-delivery replays as `agent_session_operation_unknown`, as it always has.
|
||||
export const AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY =
|
||||
'agent.launch.prompt-unconfirmed.v1' as const
|
||||
|
||||
export const AGENT_LAUNCH_RUNTIME_CAPABILITIES = [
|
||||
AGENT_LAUNCH_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_REPLAY_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_REPLAY_REQUIRED_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY
|
||||
AGENT_LAUNCH_PROMPT_CARRY_RUNTIME_CAPABILITY,
|
||||
AGENT_LAUNCH_PROMPT_UNCONFIRMED_RUNTIME_CAPABILITY
|
||||
] as const
|
||||
|
||||
Reference in New Issue
Block a user