fix(native-chat): show Claude working from the send, not the provider echo

A structured session read as working only once a turnLifecycle row existed.
Codex writes that row ~150ms after the send; Claude cannot write it until the
SDK echoes the user message back, measured at a 3.4s median and 18s at p90, so
the chat and every session list read idle for the whole wait.

The journalled submission is the host's own evidence a turn is owed, so the
shared projection reads it too. `unknown` still counts -- the ack budget
elapsing answers delivery, not whether work is owed -- while a recovered
`unknown` does not, which needed the existing row flag carried onto the
projected submission.

Claude's activity line now stays the generic fallback. Its only turn-wide frame
carries a bare token, and its task_* prose describes a spawned task rather than
this turn; compaction is kept because it explains an otherwise silent wait.
This commit is contained in:
Merge Sim
2026-09-09 18:05:48 -07:00
parent 7dd183d82d
commit 52c03acdc6
15 changed files with 282 additions and 74 deletions
@@ -11,7 +11,10 @@ import {
import { encodeNativeChatTranscriptIdentity } from '../../../src/shared/native-chat-transcript-retention'
import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send'
import { projectStructuredAgentSessionMessages } from '../../../src/shared/structured-agent-session-message-projection'
import { activeStructuredAgentSessionTurnId } from '../../../src/shared/structured-agent-session-projection'
import {
activeStructuredAgentSessionTurnId,
hasUnansweredStructuredAgentSessionDispatch
} from '../../../src/shared/structured-agent-session-projection'
import {
pendingStructuredApproval,
pendingStructuredQuestion,
@@ -291,7 +294,10 @@ export function useMobileStructuredAgentSession(args: {
loadingEarlier: loadingOlder,
loadEarlier
},
isWorking: activeStructuredAgentSessionTurnId(state.items) !== null,
// A dispatch the provider has not answered yet is already work — see the desktop hook.
isWorking:
activeStructuredAgentSessionTurnId(state.items) !== null ||
hasUnansweredStructuredAgentSessionDispatch(state.submissions),
turnId: activeStructuredAgentSessionTurnId(state.items),
sendWithOutcome,
cancel,
@@ -110,6 +110,8 @@ describe('crash between provider accept and journal commit', () => {
expect(restarted.pendingSubmissions().map((entry) => entry.clientMessageId)).toEqual(['cm_1'])
await restarted.markPendingSubmissionsUnknown(2)
expect(restarted.submissions()[0]?.dispatchState).toBe('unknown')
// Marks the send as outlived by its writer, so no reader reports it as still working.
expect(restarted.submissions()[0]?.recovered).toBe(true)
const [outcome] = reconcileSubmissions({
submissions: restarted.submissions(),
@@ -260,6 +260,9 @@ function applyDispatch(
submission.providerItemId = row.providerItemId
submission.reason = row.reason
submission.resolvedAt = row.ts
if (row.recovered) {
submission.recovered = row.recovered
}
if (row.state !== 'accepted' || !row.providerItemId) {
return
}
@@ -38,36 +38,24 @@ describe('provider frame activity', () => {
}
})
it('uses Claude descriptions and safe semantic status without exposing tool labels', () => {
expect(
claudeProviderFrameActivity('message:system:task_started', {
description: 'Trace the activity channel'
})
).toBe('Working on: Trace the activity channel')
expect(
claudeProviderFrameActivity('message:system:task_progress', {
description: 'Reading tests',
summary: 'Checking remote compatibility'
})
).toBe('Checking remote compatibility')
expect(
claudeProviderFrameActivity('message:system:task_updated', {
patch: { description: 'Validating the renderer' }
})
).toBe('Validating the renderer')
it('leaves the Claude line on the generic fallback, since Claude never narrates its turn', () => {
// Prose on these frames belongs to a spawned task, not to this turn.
for (const [kind, payload] of [
['message:system:task_started', { description: 'Trace the activity channel' }],
['message:system:task_progress', { summary: 'Checking remote compatibility' }],
['message:system:task_updated', { patch: { description: 'Validating the renderer' } }],
['message:system:control_request_progress', { status: 'api_retry' }],
['message:tool_progress', { tool_name: 'ReadSecretFile' }]
] as const) {
expect(claudeProviderFrameActivity(kind, payload)).toBeNull()
}
// `requesting` holds for nearly the whole turn and says no more than the fallback.
expect(claudeProviderFrameActivity('message:system:status', { status: 'requesting' })).toBeNull()
expect(claudeProviderFrameActivity('message:system:status', { status: 'compacting' })).toBe(
'Compacting the conversation'
)
expect(
claudeProviderFrameActivity('message:system:control_request_progress', {
status: 'api_retry'
})
).toBe('Retrying a side question')
expect(
claudeProviderFrameActivity('message:tool_progress', {
tool_name: 'ReadSecretFile'
})
).toBeNull()
// An unmodeled frame still declines to answer, so it cannot clear live copy.
expect(claudeProviderFrameActivity('message:system:unknown_frame', {})).toBeUndefined()
})
it('falls through on protocol noise and bounds long copy', () => {
@@ -100,40 +100,29 @@ export function codexProviderFrameActivity(
return itemType ? (CODEX_ITEM_ACTIVITY[itemType] ?? null) : null
}
/**
* Claude does not narrate its own turn, so the activity line stays the generic fallback.
*
* Codex names each item it starts, which is what makes its line worth reading. Claude's only
* turn-wide frame is `system/status`, whose payload is a bare token — every sentence Orca ever
* put on this line for it was Orca's own wording for `requesting`, which is true for nearly the
* whole turn and says no more than the fallback does. Its `task_*` frames do carry prose, but
* they are keyed by task id and subagent type: they describe a spawned task, not this turn, and
* the background-tasks strip already owns that. Compaction is the one exception kept — a real,
* rare state that explains an otherwise unexplained wait, and the Codex map reports it too.
*/
export function claudeProviderFrameActivity(kind: string, payload: unknown): ActivityText {
const source = record(payload)
if (kind === 'message:system:task_started') {
if (source?.ambient === true || source?.skip_transcript === true) {
return null
}
const description = providerActivityText(stringField(source, 'description'))
return description ? providerActivityText(`Working on: ${description}`) : null
}
if (kind === 'message:system:task_progress') {
return providerActivityText(
stringField(source, 'summary') ?? stringField(source, 'description')
)
}
if (kind === 'message:system:task_updated') {
return providerActivityText(stringField(record(source?.patch), 'description'))
}
if (kind === 'message:system:status') {
const status = stringField(source, 'status')
return status === 'compacting'
? 'Compacting the conversation'
: status === 'requesting'
? 'Requesting a response'
: null
return stringField(source, 'status') === 'compacting' ? 'Compacting the conversation' : null
}
if (kind === 'message:system:control_request_progress') {
const status = stringField(source, 'status')
return status === 'started'
? 'Exploring a side question'
: status === 'api_retry'
? 'Retrying a side question'
: null
}
if (kind === 'message:tool_progress') {
if (
kind === 'message:system:task_started' ||
kind === 'message:system:task_progress' ||
kind === 'message:system:task_updated' ||
kind === 'message:system:control_request_progress' ||
kind === 'message:tool_progress'
) {
return null
}
return undefined
@@ -248,10 +248,11 @@ describe('provider turn activity routing', () => {
})
)
expect(state.rows).toHaveLength(turnRows)
// Only compaction reaches the line; task and side-question prose is not this turn's work.
expect(state.activities.slice(-3)).toEqual([
{ turnId: TURN_ID, text: 'Checking the renderer state' },
null,
{ turnId: TURN_ID, text: 'Compacting the conversation' },
{ turnId: TURN_ID, text: 'Exploring a side question' }
null
])
translator.handle(claudeMessage({ type: 'tool_progress', tool_name: 'SecretReader' }))
@@ -153,6 +153,36 @@ describe('StructuredAgentSessionStatusFeed', () => {
])
})
it('publishes working from the pending submission, before the provider replays the turn', async () => {
const journal = await openJournal()
const { feed, events } = feedFor(new Map([[SESSION, { journal }]]))
events.length = 0
await journal.appendSubmission({
clientMessageId: 'client-1',
payloadFingerprint: 'fingerprint-1',
body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] },
fence: 1
})
feed.publish(SESSION)
expect(events.at(-1)).toEqual({
type: 'status',
session: expect.objectContaining({ status: 'working' })
})
await journal.resolveDispatch({
clientMessageId: 'client-1',
state: 'accepted',
providerIdentity: USER_IDENTITY,
fence: 1
})
feed.publish(SESSION)
expect(events.at(-1)).toEqual({
type: 'status',
session: expect.objectContaining({ status: 'idle' })
})
})
it('publishes working, then idle once the running marker is tombstoned, and never a repeat', async () => {
const journal = await openJournal()
const { feed, events } = feedFor(new Map([[SESSION, { journal }]]))
@@ -153,7 +153,9 @@ export class StructuredAgentSessionStatusFeed {
journal: AgentSessionJournal
): AgentSessionStatusSummary {
// An unreadable journal projects as "no turn": the chat itself shows the reset.
const items = journal.isReadOnly ? [] : journal.snapshot().items
const snapshot = journal.isReadOnly ? null : journal.snapshot()
const items = snapshot?.items ?? []
const submissions = snapshot?.submissions ?? []
const record = this.deps.getRecord(sessionId)
const providerSession = structuredAgentSessionProviderSessionMetadata(record)
// The journal has no model: the record's acknowledged options are where an owner
@@ -164,7 +166,7 @@ export class StructuredAgentSessionStatusFeed {
workspaceId: session.params.location.workspaceId,
agent: session.params.provider,
...(session.hasProviderChild ? { hostExecutionOwned: true as const } : {}),
...projectStructuredAgentSessionStatusSummary(items),
...projectStructuredAgentSessionStatusSummary(items, submissions),
...(record?.rewind?.phase === 'prepared' || record?.rewind?.phase === 'provider-succeeded'
? { rewindBlockedReason: 'outcome-unknown' as const }
: {}),
@@ -352,7 +352,9 @@ export function NativeChatStructuredSession(
targetPtyId={null}
agent={props.agent}
canSend={!prompt}
isWorking={controller.isWorking}
// Stop, not status: only a provider-minted turn can be interrupted, so the button
// must not flip while a dispatch is still unanswered.
isWorking={controller.turnId !== null}
onStop={() => {
if (controller.turnId) {
void controller.cancel(controller.turnId)
@@ -10,6 +10,7 @@ const mocks = vi.hoisted(() => ({
}))
let fence = 3
let sessionCommands: { name: string; kind: 'command' | 'skill' }[] | undefined
let submissions: AgentJournalSubmission[] = []
vi.mock('@/runtime/structured-agent-session-client', () => ({
callStructuredAgentSession: mocks.call
@@ -25,7 +26,7 @@ vi.mock('./use-structured-agent-session-read', () => ({
fence,
commands: sessionCommands,
items: [],
submissions: [],
submissions,
status: 'ready',
error: null,
hasOlder: false,
@@ -47,6 +48,7 @@ vi.mock('./use-structured-agent-session-outbox', () => ({
})
}))
import type { AgentJournalSubmission } from '../../../../shared/agent-session-journal-types'
import {
applyNativeChatSessionOptionSettingsMutation,
resolveStructuredLaunchSeedOptions
@@ -95,10 +97,72 @@ const OPTIONS = {
current: { model: 'gpt-live', effort: 'medium' }
}
describe('useStructuredAgentSession working state', () => {
beforeEach(() => {
vi.clearAllMocks()
fence = 3
submissions = []
mocks.call.mockResolvedValue(null)
})
it('reports work from an unanswered dispatch, and keeps the turn id provider-minted', () => {
submissions = [
{
clientMessageId: 'client-1',
fence: 3,
payloadFingerprint: 'fingerprint-1',
dispatchState: 'pending',
providerItemId: null,
reason: null,
submittedAt: 1,
resolvedAt: null
}
]
const { result } = renderHook(() =>
useStructuredAgentSession({
sessionId: 'session-1',
agent: 'codex',
target: LOCAL_TARGET,
isVisible: true
})
)
expect(result.current.isWorking).toBe(true)
// Only the provider can mint a cancellable turn, so Stop stays unavailable here.
expect(result.current.turnId).toBeNull()
})
it('reports no work once the dispatch resolves and no turn is running', () => {
submissions = [
{
clientMessageId: 'client-1',
fence: 3,
payloadFingerprint: 'fingerprint-1',
dispatchState: 'accepted',
providerItemId: 'codex:thread-1:turn-1',
reason: null,
submittedAt: 1,
resolvedAt: 2
}
]
const { result } = renderHook(() =>
useStructuredAgentSession({
sessionId: 'session-1',
agent: 'codex',
target: LOCAL_TARGET,
isVisible: true
})
)
expect(result.current.isWorking).toBe(false)
})
})
describe('useStructuredAgentSession options', () => {
beforeEach(() => {
vi.clearAllMocks()
fence = 3
submissions = []
mocks.operationId
.mockReset()
.mockReturnValueOnce('operation-1')
@@ -22,7 +22,10 @@ import {
structuredAgentSessionOptionPicks,
structuredAgentSessionOptionSnapshot
} from '../../../../shared/structured-agent-session-options'
import { activeStructuredAgentSessionTurnId } from '../../../../shared/structured-agent-session-projection'
import {
activeStructuredAgentSessionTurnId,
hasUnansweredStructuredAgentSessionDispatch
} from '../../../../shared/structured-agent-session-projection'
import type { RuntimeClientTarget } from '@/runtime/runtime-rpc-client'
import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client'
import { useStructuredAgentSessionHold } from './use-structured-agent-session-hold'
@@ -80,6 +83,11 @@ export function useStructuredAgentSession(args: {
// Refresh options each turn to confirm which model the provider actually selected.
const turnId = activeStructuredAgentSessionTurnId(state.items)
// A dispatch the provider has not answered is already work; Claude's running row trails the
// send by seconds, and only a provider-minted turn is cancellable, so the two stay separate.
const isWorking =
turnId !== null ||
hasUnansweredStructuredAgentSessionDispatch(state.submissions)
const turnActivity = useMemo(
() => selectStructuredAgentTurnActivity(state.items, turnId, state.activity),
[state.activity, state.items, turnId]
@@ -210,7 +218,7 @@ export function useStructuredAgentSession(args: {
send: (...input: Parameters<typeof outboxController.send>) =>
!commandPending.current && outboxController.send(...input),
retry: outboxController.retry,
isWorking: turnId !== null,
isWorking,
turnActivity,
...backgroundTasksView,
turnId,
+2 -1
View File
@@ -186,7 +186,8 @@ export const AgentJournalSubmissionSchema = z.object({
providerItemId: z.string().nullable(),
reason: z.string().nullable(),
submittedAt: z.number(),
resolvedAt: z.number().nullable()
resolvedAt: z.number().nullable(),
recovered: z.literal(true).optional()
})
export function isAdmissibleAgentJournalItemBody(value: unknown): value is AgentJournalItemBody {
@@ -202,6 +202,9 @@ export type AgentJournalSubmission = {
reason: string | null
submittedAt: number
resolvedAt: number | null
/** Set when crash reconciliation resolved the dispatch, not the provider. A live
* `unknown` is a send still outstanding; a recovered one outlived its writer. */
recovered?: true
}
/** Durable answer to "did my send land?", keyed by client message id. Only an
@@ -1,10 +1,14 @@
import { describe, expect, it } from 'vitest'
import { AGENT_STATUS_MAX_FIELD_LENGTH } from './agent-status-field-normalization'
import type { AgentJournalRenderItem } from './agent-session-journal-types'
import type {
AgentJournalRenderItem,
AgentJournalSubmission
} from './agent-session-journal-types'
import { parsePaneKey } from './stable-pane-id'
import {
activeStructuredAgentSessionTurnId,
hasPersistedStructuredAgentSessionTurn,
hasUnansweredStructuredAgentSessionDispatch,
projectStructuredItemToNativeChat,
projectStructuredAgentSessionStatus,
projectStructuredAgentSessionStatusSummary,
@@ -19,6 +23,22 @@ function item(
return { itemId, sequence, revision: 1, observedAt: sequence, body }
}
function submission(
clientMessageId: string,
dispatchState: AgentJournalSubmission['dispatchState']
): AgentJournalSubmission {
return {
clientMessageId,
fence: 1,
payloadFingerprint: clientMessageId,
dispatchState,
providerItemId: null,
reason: null,
submittedAt: 1,
resolvedAt: dispatchState === 'pending' ? null : 2
}
}
describe('structured agent session status projection', () => {
it('reuses immutable item projections and refreshes revisions and resolved prompts', () => {
const original = item('diff', 1, {
@@ -139,6 +159,61 @@ describe('structured agent session status projection', () => {
})
})
it('reads a session as working while a dispatch is unanswered, before any lifecycle row', () => {
const asked = item('asked', 1, {
kind: 'message',
role: 'user',
blocks: [{ type: 'text', text: 'go' }]
})
const pending = [submission('m1', 'pending')]
expect(hasUnansweredStructuredAgentSessionDispatch(pending)).toBe(true)
expect(projectStructuredAgentSessionStatus([asked], pending)).toBe('working')
// The first send has no journalled message until the provider replays it.
expect(projectStructuredAgentSessionStatusSummary([], pending)).toEqual({
status: 'working',
latestPrompt: ''
})
expect(projectStructuredAgentSessionStatusSummary([asked], pending)).toEqual({
status: 'working',
latestPrompt: 'go'
})
})
it('stops reading a resolved dispatch as work, and lets a pending prompt outrank it', () => {
const asked = item('asked', 1, {
kind: 'message',
role: 'user',
blocks: [{ type: 'text', text: 'go' }]
})
const prompt = item('prompt', 2, {
kind: 'approval',
title: 'Run command?',
detail: null,
options: [{ id: 'yes', label: 'Allow' }],
resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null }
})
for (const state of ['accepted', 'rejected'] as const) {
expect(hasUnansweredStructuredAgentSessionDispatch([submission('m1', state)])).toBe(false)
expect(projectStructuredAgentSessionStatus([asked], [submission('m1', state)])).toBe('idle')
}
// The ack budget elapsing is a delivery answer, not an answer about the turn.
expect(hasUnansweredStructuredAgentSessionDispatch([submission('m1', 'unknown')])).toBe(true)
expect(
hasUnansweredStructuredAgentSessionDispatch([
{ ...submission('m1', 'unknown'), recovered: true }
])
).toBe(false)
expect(projectStructuredAgentSessionStatus([asked, prompt], [submission('m1', 'pending')])).toBe(
'attention'
)
expect(projectStructuredAgentSessionStatusSummary([], [])).toEqual({
status: null,
latestPrompt: ''
})
})
it('carries the running tool and the newest assistant prose the sidebar row shows', () => {
const ask = item('ask', 1, {
kind: 'message',
@@ -5,6 +5,7 @@ import {
} from './agent-status-field-normalization'
import type {
AgentJournalRenderItem,
AgentJournalSubmission,
AgentJournalToolCallItem
} from './agent-session-journal-types'
import {
@@ -175,6 +176,29 @@ export function hasPersistedStructuredAgentSessionTurn(
)
}
/**
* A send the host has journaled that the provider has neither opened a turn for nor refused.
*
* Codex declares `turn/started` within ~150ms, but Claude's running row can only be written once
* the SDK echoes the user message back — a 3.4s median and 18s at p90 on real journals. Waiting
* on that echo to call a session working leaves the whole gap reading idle in the chat and in
* every session list, so the send itself is the evidence.
*
* `unknown` still counts: it only means the ack budget elapsed, which happens on 30% of Claude
* sends whose turn then arrives anyway, and delivery confidence is a separate question from
* whether work is owed. A recovered `unknown` does not — that one outlived the host generation
* that sent it, so there is nothing still running to report.
*/
export function hasUnansweredStructuredAgentSessionDispatch(
submissions: readonly AgentJournalSubmission[]
): boolean {
return submissions.some(
(submission) =>
submission.dispatchState === 'pending' ||
(submission.dispatchState === 'unknown' && submission.recovered !== true)
)
}
export type StructuredAgentSessionProjectedStatus = 'working' | 'attention' | 'idle'
export function structuredAgentSessionTabId(sessionId: string): string {
@@ -182,7 +206,8 @@ export function structuredAgentSessionTabId(sessionId: string): string {
}
export function projectStructuredAgentSessionStatus(
items: readonly AgentJournalRenderItem[]
items: readonly AgentJournalRenderItem[],
submissions: readonly AgentJournalSubmission[] = []
): StructuredAgentSessionProjectedStatus {
if (
items.some(
@@ -193,7 +218,10 @@ export function projectStructuredAgentSessionStatus(
) {
return 'attention'
}
return activeStructuredAgentSessionTurnId(items) ? 'working' : 'idle'
return activeStructuredAgentSessionTurnId(items) ||
hasUnansweredStructuredAgentSessionDispatch(submissions)
? 'working'
: 'idle'
}
function messageProse(blocks: readonly NativeChatBlock[]): string {
@@ -270,12 +298,18 @@ export type StructuredAgentSessionStatusProjection = {
* the 8 KB body): a streamed reply re-projects on every journal checkpoint, so the frame
* has to stay small even though the row only ever renders one line of it. */
export function projectStructuredAgentSessionStatusSummary(
items: readonly AgentJournalRenderItem[]
items: readonly AgentJournalRenderItem[],
submissions: readonly AgentJournalSubmission[] = []
): StructuredAgentSessionStatusProjection {
if (!hasPersistedStructuredAgentSessionTurn(items)) {
// A first send has no journalled message until the provider replays it, so the pending
// dispatch is also what makes a brand-new session listable at all.
if (
!hasPersistedStructuredAgentSessionTurn(items) &&
!hasUnansweredStructuredAgentSessionDispatch(submissions)
) {
return { status: null, latestPrompt: '' }
}
const status = projectStructuredAgentSessionStatus(items)
const status = projectStructuredAgentSessionStatus(items, submissions)
const activeToolCall = status === 'working' ? activeStructuredAgentSessionToolCall(items) : null
const toolName = activeToolCall
? normalizeOptionalField(activeToolCall.name, AGENT_STATUS_TOOL_NAME_MAX_LENGTH)