mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 00:02:41 +00:00
fix(native-chat): keep a structured agent's tool line between tool calls (#22349)
* fix(native-chat): keep a structured agent's tool line between tool calls A structured session's status named a tool only while the call was still running, so the sidebar's tool line went blank whenever the agent was thinking or writing between calls. Terminal agents keep naming the finished tool until the next one starts, and clear it after a failure. The structured status projection now does the same: a running call wins, otherwise the turn's newest root call if it completed. * fix(native-chat): bound the structured tool line by the running turn, not the user row A send made while a turn is running writes its user row into the journal straight away, and the turn keeps going. Stopping the scan at that row blanked the tool line while a tool was still running. The scan now runs to the turn record and names a call only when that record is still running, so a turn that already ended never lends its last tool to a pending follow-up. This lookup was the running-only lookup's only production caller, so it replaces that lookup instead of sitting beside it. * fix(native-chat): keep naming a structured agent's failed tool until the next one Clearing the tool line after a failed call brought the blank gap back for much of a turn: Codex marks any nonzero exit as failed, so a search with no match or a red test run is enough. The failure already shows on the tool's own row in the transcript. The running turn's newest running call still wins; otherwise its newest root call is named whatever it settled to. * fix(native-chat): name a structured Codex edit on the tool line as the chat draws it Once a Codex edit's changes exist, its apply_patch call becomes a diff row, which the status lookup skipped, so the row named the command before the edit. The chat's tool-call block for a journal row now comes from one shared builder, and the status lookup reads the same definition: a diff is named as Diff with its path, and counts as settled since it carries no lifecycle. * docs(native-chat): describe the structured tool field as running-or-latest The status summary's toolName/toolInput now name the running turn's latest tool between calls, not only a running one. Update the wire type and status bridge comments that still said "the running tool".
This commit is contained in:
@@ -22,7 +22,7 @@ import {
|
||||
projectStructuredAgentSessionStatus,
|
||||
projectStructuredAgentSessionStatusSummary
|
||||
} from '../../shared/structured-agent-session-projection'
|
||||
import { activeStructuredAgentSessionToolCall } from '../../shared/structured-agent-session-live-turn'
|
||||
import { statusStructuredAgentSessionToolCall } from '../../shared/structured-agent-session-live-turn'
|
||||
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
|
||||
import { createClaudeJournalTranslator } from './claude-structured-journal-translation'
|
||||
|
||||
@@ -268,7 +268,7 @@ describe('a Claude turn the provider resumed on its own', () => {
|
||||
)
|
||||
|
||||
expect(projected(items())).toBe('working')
|
||||
expect(activeStructuredAgentSessionToolCall(items())?.name).toBe('Bash')
|
||||
expect(statusStructuredAgentSessionToolCall(items())?.name).toBe('Bash')
|
||||
expect(projectStructuredAgentSessionStatusSummary(items(), [], null).toolName).toBe('Bash')
|
||||
})
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@ function projectStatus(
|
||||
prompt: summary.latestPrompt,
|
||||
agentType: tab.agentSessionAgent,
|
||||
// The host projects these from the journal so the row reads like a hook-reported one:
|
||||
// the running tool while a turn is live, the agent's last words once it settles.
|
||||
// the turn's running or latest tool while it is live, the agent's last words once it settles.
|
||||
...(summary.model ? { model: summary.model } : {}),
|
||||
...(summary.toolName ? { toolName: summary.toolName } : {}),
|
||||
...(summary.toolInput ? { toolInput: summary.toolInput } : {}),
|
||||
|
||||
@@ -212,7 +212,8 @@ export type AgentSessionStatusSummary = {
|
||||
latestPrompt: string
|
||||
/** Provider model in force for the next turn; absent until the host has read the options. */
|
||||
model?: string
|
||||
/** The tool the running turn is inside. Absent unless `status` is 'working'. */
|
||||
/** The tool the running turn is inside, else the last one it used. Absent unless `status`
|
||||
* is 'working'. */
|
||||
toolName?: string
|
||||
toolInput?: string
|
||||
/** Preview of the newest assistant prose, so a settled row says what the agent said. */
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import type { AgentJournalRenderItem } from './agent-session-journal-types'
|
||||
import {
|
||||
activeStructuredAgentSessionToolCall,
|
||||
statusStructuredAgentSessionToolCall,
|
||||
isStructuredAgentSessionThinking
|
||||
} from './structured-agent-session-live-turn'
|
||||
|
||||
@@ -159,7 +159,7 @@ describe("the live-turn readers answer for the session's own agent", () => {
|
||||
input: { pattern: 'x' },
|
||||
state: 'running'
|
||||
})
|
||||
expect(activeStructuredAgentSessionToolCall([turnStart, spawnCall, childCall])?.name).toBe(
|
||||
expect(statusStructuredAgentSessionToolCall([turnStart, spawnCall, childCall])?.name).toBe(
|
||||
'Task'
|
||||
)
|
||||
})
|
||||
@@ -171,7 +171,7 @@ describe("the live-turn readers answer for the session's own agent", () => {
|
||||
input: { pattern: 'x' },
|
||||
state: 'running'
|
||||
})
|
||||
expect(activeStructuredAgentSessionToolCall([turnStart, childCall])).toBeNull()
|
||||
expect(statusStructuredAgentSessionToolCall([turnStart, childCall])).toBeNull()
|
||||
})
|
||||
|
||||
it('treats an agent id that failed to resolve as a child, not as the parent', () => {
|
||||
@@ -184,7 +184,7 @@ describe("the live-turn readers answer for the session's own agent", () => {
|
||||
{ kind: 'tool-call', name: 'Grep', input: { pattern: 'x' }, state: 'running' },
|
||||
''
|
||||
)
|
||||
expect(activeStructuredAgentSessionToolCall([turnStart, spawnCall, unresolved])?.name).toBe(
|
||||
expect(statusStructuredAgentSessionToolCall([turnStart, spawnCall, unresolved])?.name).toBe(
|
||||
'Task'
|
||||
)
|
||||
})
|
||||
@@ -197,7 +197,7 @@ describe("the live-turn readers answer for the session's own agent", () => {
|
||||
state: 'running'
|
||||
})
|
||||
expect(
|
||||
activeStructuredAgentSessionToolCall([turnStart, spawnCall, legacyChildCall])?.name
|
||||
statusStructuredAgentSessionToolCall([turnStart, spawnCall, legacyChildCall])?.name
|
||||
).toBe('Grep')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -18,11 +18,17 @@
|
||||
|
||||
import type {
|
||||
AgentJournalRenderItem,
|
||||
AgentJournalToolCallItem,
|
||||
AgentJournalTurnLifecycle
|
||||
} from './agent-session-journal-types'
|
||||
import { isRootAgentJournalItem } from './agent-session-journal-producer'
|
||||
import { readAgentJournalTurn } from './agent-session-turn-record'
|
||||
import type { NativeChatToolCallBlock } from './native-chat-types'
|
||||
import {
|
||||
isRunningStructuredAgentSessionToolAction,
|
||||
isStructuredAgentSessionToolAction,
|
||||
structuredAgentSessionToolCallBlock,
|
||||
type StructuredAgentSessionToolAction
|
||||
} from './structured-agent-session-tool-call-block'
|
||||
|
||||
export function activeStructuredAgentSessionTurnId(
|
||||
items: readonly AgentJournalRenderItem[]
|
||||
@@ -122,21 +128,30 @@ export function isStructuredAgentSessionThinking(
|
||||
return false
|
||||
}
|
||||
|
||||
/** The tool call the SESSION'S OWN agent is still inside, or null when nothing is
|
||||
* running. An abandoned `running` call from an earlier crashed turn can never be
|
||||
* reported as live work, and neither can a subagent's — while a child runs a
|
||||
* tool, the parent is still inside the call that spawned it. */
|
||||
export function activeStructuredAgentSessionToolCall(
|
||||
/** The tool the status row names for the SESSION'S OWN agent, as the chat draws it: the running
|
||||
* turn's newest running call, else its newest tool action whatever it settled to, so the line
|
||||
* never blanks mid-turn. Nothing is named unless the scan reaches a RUNNING turn record, so an
|
||||
* ended turn's calls never surface; a mid-turn send's user row is not a boundary. */
|
||||
export function statusStructuredAgentSessionToolCall(
|
||||
items: readonly AgentJournalRenderItem[]
|
||||
): AgentJournalToolCallItem | null {
|
||||
): NativeChatToolCallBlock | null {
|
||||
let newest: StructuredAgentSessionToolAction | null = null
|
||||
let running: StructuredAgentSessionToolAction | null = null
|
||||
for (let index = items.length - 1; index >= 0; index -= 1) {
|
||||
const item = items[index]
|
||||
const body = item?.body
|
||||
if (readAgentJournalTurn(body)) {
|
||||
return null
|
||||
const turn = readAgentJournalTurn(body)
|
||||
if (turn) {
|
||||
const named = turn.state === 'running' ? (running ?? newest) : null
|
||||
// Built only for the winner: the host re-projects this on every journal change.
|
||||
return named ? structuredAgentSessionToolCallBlock(named) : null
|
||||
}
|
||||
if (body?.kind === 'tool-call' && body.state === 'running' && isRootAgentJournalItem(item)) {
|
||||
return body
|
||||
if (running || !isStructuredAgentSessionToolAction(body) || !isRootAgentJournalItem(item)) {
|
||||
continue
|
||||
}
|
||||
newest ??= body
|
||||
if (isRunningStructuredAgentSessionToolAction(body)) {
|
||||
running = body
|
||||
}
|
||||
}
|
||||
return null
|
||||
|
||||
@@ -9,11 +9,11 @@ import {
|
||||
projectStructuredItemToNativeChat,
|
||||
projectStructuredItemsToNativeChat,
|
||||
latestStructuredAgentSessionAssistantMessage,
|
||||
activeStructuredAgentSessionToolCall,
|
||||
projectStructuredAgentSessionStatus,
|
||||
projectStructuredAgentSessionStatusSummary,
|
||||
structuredAgentSessionPaneKey
|
||||
} from './structured-agent-session-projection'
|
||||
import { statusStructuredAgentSessionToolCall } from './structured-agent-session-live-turn'
|
||||
|
||||
function item(
|
||||
itemId: string,
|
||||
@@ -74,6 +74,10 @@ describe('structured agent session status projection', () => {
|
||||
timestamp: 2000,
|
||||
blocks: [{ type: 'tool-call' }, { type: 'tool-result', output: '@@\n+second' }]
|
||||
})
|
||||
expect(second?.blocks).toEqual([
|
||||
{ type: 'tool-call', name: 'Diff', input: { path: 'a.ts' } },
|
||||
{ type: 'tool-result', output: '@@\n+second' }
|
||||
])
|
||||
const pending = item('approval', 2, {
|
||||
kind: 'approval',
|
||||
title: 'Allow?',
|
||||
@@ -560,7 +564,7 @@ describe("producer linkage — a subagent's output never speaks for the parent",
|
||||
|
||||
it("shows the parent's own prose and its own running call, not the child's newer ones", () => {
|
||||
expect(latestStructuredAgentSessionAssistantMessage(items)).toBe('delegating')
|
||||
expect(activeStructuredAgentSessionToolCall(items)?.name).toBe('Task')
|
||||
expect(statusStructuredAgentSessionToolCall(items)?.name).toBe('Task')
|
||||
})
|
||||
|
||||
it("publishes the parent's own line and call on the summary the sidebar reads", () => {
|
||||
@@ -573,6 +577,25 @@ describe("producer linkage — a subagent's output never speaks for the parent",
|
||||
expect(summary.toolInput).toBeTruthy()
|
||||
})
|
||||
|
||||
it("names the parent's own settled call, not a child's newer settled one", () => {
|
||||
const parentRead = item('root-read', 4, {
|
||||
kind: 'tool-call',
|
||||
name: 'Read',
|
||||
input: { file_path: '/repo/a.ts' },
|
||||
state: 'completed'
|
||||
})
|
||||
const childFailed = childItem('child-grep', 6, {
|
||||
kind: 'tool-call',
|
||||
name: 'Grep',
|
||||
input: { pattern: 'x' },
|
||||
state: 'failed'
|
||||
})
|
||||
expect(
|
||||
projectStructuredAgentSessionStatusSummary([userAsk, turnRunning, parentRead, childFailed])
|
||||
.toolName
|
||||
).toBe('Read')
|
||||
})
|
||||
|
||||
it("still renders the child's output in the transcript", () => {
|
||||
// The other direction: scoping the STATUS readers must not delete subagent
|
||||
// output from the chat.
|
||||
@@ -603,7 +626,7 @@ describe("producer linkage — a subagent's output never speaks for the parent",
|
||||
// reachable today — the point is that the rule does not rely on that.)
|
||||
const windowed = [childProse, childCall]
|
||||
expect(latestStructuredAgentSessionAssistantMessage(windowed)).toBe('')
|
||||
expect(activeStructuredAgentSessionToolCall(windowed)).toBeNull()
|
||||
expect(statusStructuredAgentSessionToolCall(windowed)).toBeNull()
|
||||
})
|
||||
|
||||
it("does not quote a subagent's own user-role prompt as the session's", () => {
|
||||
|
||||
@@ -11,16 +11,19 @@ import {
|
||||
} from './agent-status-types'
|
||||
import { describeToolInput } from './native-chat-tool-summary'
|
||||
import {
|
||||
activeStructuredAgentSessionToolCall,
|
||||
activeStructuredAgentSessionTurnId
|
||||
activeStructuredAgentSessionTurnId,
|
||||
statusStructuredAgentSessionToolCall
|
||||
} from './structured-agent-session-live-turn'
|
||||
import {
|
||||
isStructuredAgentSessionToolAction,
|
||||
structuredAgentSessionToolCallBlock
|
||||
} from './structured-agent-session-tool-call-block'
|
||||
|
||||
import type { NativeChatBlock, NativeChatMessage } from './native-chat-types'
|
||||
import { sha256 } from './sha256'
|
||||
|
||||
// Re-exported so the live-turn readers' existing consumers keep one import site.
|
||||
export {
|
||||
activeStructuredAgentSessionToolCall,
|
||||
activeStructuredAgentSessionTurnId,
|
||||
newestStructuredAgentSessionTurn
|
||||
} from './structured-agent-session-live-turn'
|
||||
@@ -53,23 +56,18 @@ function itemBlocks(item: AgentJournalRenderItem): {
|
||||
if (body.kind === 'message') {
|
||||
return { role: body.role, blocks: body.blocks }
|
||||
}
|
||||
if (body.kind === 'tool-call') {
|
||||
if (isStructuredAgentSessionToolAction(body)) {
|
||||
const call = structuredAgentSessionToolCallBlock(body)
|
||||
if (body.kind === 'diff') {
|
||||
return {
|
||||
role: 'assistant',
|
||||
blocks: [call, { type: 'tool-result', output: boundedText(body.patch) }]
|
||||
}
|
||||
}
|
||||
return {
|
||||
role: 'assistant',
|
||||
blocks: [
|
||||
{
|
||||
type: 'tool-call',
|
||||
name: body.name,
|
||||
input: body.input,
|
||||
state: body.state,
|
||||
...(body.callId !== undefined ? { callId: body.callId } : {}),
|
||||
...(body.mcpIdentity !== undefined ? { mcpIdentity: body.mcpIdentity } : {}),
|
||||
...(body.exitCode !== undefined ? { exitCode: body.exitCode } : {}),
|
||||
...(body.durationMs !== undefined ? { durationMs: body.durationMs } : {}),
|
||||
...(body.webSearchResults !== undefined
|
||||
? { webSearchResults: body.webSearchResults }
|
||||
: {})
|
||||
},
|
||||
call,
|
||||
...(body.output
|
||||
? [
|
||||
{
|
||||
@@ -82,15 +80,6 @@ function itemBlocks(item: AgentJournalRenderItem): {
|
||||
]
|
||||
}
|
||||
}
|
||||
if (body.kind === 'diff') {
|
||||
return {
|
||||
role: 'assistant',
|
||||
blocks: [
|
||||
{ type: 'tool-call', name: 'Diff', input: { path: body.path } },
|
||||
{ type: 'tool-result', output: boundedText(body.patch) }
|
||||
]
|
||||
}
|
||||
}
|
||||
if (body.kind === 'approval') {
|
||||
if (body.resolution.state === 'pending') {
|
||||
return null
|
||||
@@ -329,13 +318,13 @@ export function projectStructuredAgentSessionStatusSummary(
|
||||
return { status: null, latestPrompt: '' }
|
||||
}
|
||||
const status = projectStructuredAgentSessionStatus(items, submissions, currentFence)
|
||||
const activeToolCall = status === 'working' ? activeStructuredAgentSessionToolCall(items) : null
|
||||
const toolName = activeToolCall
|
||||
? normalizeOptionalField(activeToolCall.name, AGENT_STATUS_TOOL_NAME_MAX_LENGTH)
|
||||
const statusToolCall = status === 'working' ? statusStructuredAgentSessionToolCall(items) : null
|
||||
const toolName = statusToolCall
|
||||
? normalizeOptionalField(statusToolCall.name, AGENT_STATUS_TOOL_NAME_MAX_LENGTH)
|
||||
: undefined
|
||||
const toolInput = activeToolCall
|
||||
const toolInput = statusToolCall
|
||||
? normalizeOptionalField(
|
||||
describeToolInput(activeToolCall.input),
|
||||
describeToolInput(statusToolCall.input),
|
||||
AGENT_STATUS_TOOL_INPUT_MAX_LENGTH
|
||||
)
|
||||
: undefined
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import type {
|
||||
AgentJournalRenderItem,
|
||||
AgentJournalSubmission,
|
||||
AgentJournalToolCallState
|
||||
} from './agent-session-journal-types'
|
||||
import { describeToolInput } from './native-chat-tool-summary'
|
||||
import {
|
||||
projectStructuredAgentSessionStatusSummary,
|
||||
projectStructuredItemsToNativeChat
|
||||
} from './structured-agent-session-projection'
|
||||
import { statusStructuredAgentSessionToolCall } from './structured-agent-session-live-turn'
|
||||
|
||||
function item(
|
||||
itemId: string,
|
||||
sequence: number,
|
||||
body: AgentJournalRenderItem['body']
|
||||
): AgentJournalRenderItem {
|
||||
return { itemId, sequence, revision: 1, observedAt: sequence, body }
|
||||
}
|
||||
|
||||
function child(base: AgentJournalRenderItem): AgentJournalRenderItem {
|
||||
return { ...base, agentId: 'task-1', producerKind: 'agent' }
|
||||
}
|
||||
|
||||
function submission(clientMessageId: string): AgentJournalSubmission {
|
||||
return {
|
||||
clientMessageId,
|
||||
fence: 1,
|
||||
payloadFingerprint: clientMessageId,
|
||||
dispatchState: 'pending',
|
||||
providerItemId: null,
|
||||
reason: null,
|
||||
submittedAt: 1,
|
||||
resolvedAt: null
|
||||
}
|
||||
}
|
||||
|
||||
describe('tool line between tool calls', () => {
|
||||
const ask = item('ask', 1, {
|
||||
kind: 'message',
|
||||
role: 'user',
|
||||
blocks: [{ type: 'text', text: 'go' }]
|
||||
})
|
||||
const running = item('running', 2, {
|
||||
kind: 'status',
|
||||
text: 'Working',
|
||||
turnLifecycle: { turnId: 'turn-1', state: 'running' }
|
||||
})
|
||||
const call = (id: string, sequence: number, name: string, state: AgentJournalToolCallState) =>
|
||||
item(id, sequence, {
|
||||
kind: 'tool-call',
|
||||
name,
|
||||
input: { file_path: `/repo/${name}.ts` },
|
||||
state
|
||||
})
|
||||
|
||||
it('keeps naming the finished tool while the agent thinks', () => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
call('read', 3, 'Read', 'completed')
|
||||
])
|
||||
expect(summary).toMatchObject({ toolName: 'Read', toolInput: '/repo/Read.ts' })
|
||||
})
|
||||
|
||||
// Codex marks any nonzero exit failed (a no-match search, a red test), so clearing would blank the line.
|
||||
it('keeps naming a failed call until the next tool starts', () => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
call('read', 3, 'Read', 'completed'),
|
||||
call('edit', 4, 'Edit', 'failed')
|
||||
])
|
||||
expect(summary).toMatchObject({ toolName: 'Edit', toolInput: '/repo/Edit.ts' })
|
||||
})
|
||||
|
||||
it('prefers a running call over a newer finished one', () => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
call('bash', 3, 'Bash', 'running'),
|
||||
call('read', 4, 'Read', 'completed')
|
||||
])
|
||||
expect(summary.toolName).toBe('Bash')
|
||||
})
|
||||
|
||||
it("never carries an earlier turn's finished tool into the live one", () => {
|
||||
const nextTurn = item('next-turn', 4, {
|
||||
kind: 'status',
|
||||
text: 'Working',
|
||||
turnLifecycle: { turnId: 'turn-2', state: 'running' }
|
||||
})
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
call('read', 3, 'Read', 'completed'),
|
||||
nextTurn
|
||||
])
|
||||
expect(summary.toolName).toBeUndefined()
|
||||
})
|
||||
|
||||
// A send's user row lands at submit time, mid-turn too; the turn record bounds the turn.
|
||||
const followUp = item('follow-up', 5, {
|
||||
kind: 'message',
|
||||
role: 'user',
|
||||
blocks: [{ type: 'text', text: 'also check the tests' }]
|
||||
})
|
||||
|
||||
it("keeps naming the running turn's tool past a mid-turn send", () => {
|
||||
const pending = [submission('follow-up')]
|
||||
expect(
|
||||
projectStructuredAgentSessionStatusSummary(
|
||||
[ask, running, call('bash', 3, 'Bash', 'running'), followUp],
|
||||
pending
|
||||
).toolName
|
||||
).toBe('Bash')
|
||||
expect(
|
||||
projectStructuredAgentSessionStatusSummary(
|
||||
[ask, running, call('read', 3, 'Read', 'completed'), followUp],
|
||||
pending
|
||||
).toolName
|
||||
).toBe('Read')
|
||||
})
|
||||
|
||||
it('names nothing from an ended turn while the next send is pending', () => {
|
||||
// The record keeps its creation slot when revised to completed, so it sits before its calls.
|
||||
const ended = item('running', 2, {
|
||||
kind: 'status',
|
||||
text: 'Done',
|
||||
turnLifecycle: { turnId: 'turn-1', state: 'completed' }
|
||||
})
|
||||
const summary = projectStructuredAgentSessionStatusSummary(
|
||||
[ask, ended, call('read', 3, 'Read', 'completed'), followUp],
|
||||
[submission('follow-up')]
|
||||
)
|
||||
expect(summary.status).toBe('working')
|
||||
expect(summary.toolName).toBeUndefined()
|
||||
})
|
||||
|
||||
const patch = { head: '@@\n+x', digest: 'd', byteLength: 5, truncated: false }
|
||||
|
||||
describe('a Codex edit, which the chat draws as a Diff', () => {
|
||||
const command = (
|
||||
id: string,
|
||||
sequence: number,
|
||||
text: string,
|
||||
state: AgentJournalToolCallState
|
||||
) => item(id, sequence, { kind: 'tool-call', name: 'shell', input: { command: text }, state })
|
||||
const diff = (id: string, sequence: number, path: string) =>
|
||||
item(id, sequence, { kind: 'diff', path, patch })
|
||||
const rg = command('rg', 3, 'rg foo', 'completed')
|
||||
|
||||
it('names the edit, not the command before it', () => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
rg,
|
||||
diff('edit', 4, 'src/a.ts')
|
||||
])
|
||||
expect(summary).toMatchObject({ toolName: 'Diff', toolInput: 'src/a.ts' })
|
||||
})
|
||||
|
||||
it('moves from the running apply_patch call to the Diff the same item becomes', () => {
|
||||
const applying = item('edit', 4, {
|
||||
kind: 'tool-call',
|
||||
name: 'apply_patch',
|
||||
input: { changes: [] },
|
||||
state: 'running'
|
||||
})
|
||||
const written = { ...diff('edit', 4, 'src/a.ts'), revision: 2 }
|
||||
expect(
|
||||
projectStructuredAgentSessionStatusSummary([ask, running, rg, applying]).toolName
|
||||
).toBe('apply_patch')
|
||||
expect(projectStructuredAgentSessionStatusSummary([ask, running, rg, written])).toMatchObject(
|
||||
{ toolName: 'Diff', toolInput: 'src/a.ts' }
|
||||
)
|
||||
})
|
||||
|
||||
it('lets an older running command beat a newer Diff, which has no lifecycle', () => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
command('test', 3, 'pnpm test', 'running'),
|
||||
diff('edit', 4, 'src/a.ts')
|
||||
])
|
||||
expect(summary).toMatchObject({ toolName: 'shell', toolInput: 'pnpm test' })
|
||||
})
|
||||
|
||||
it.each(['completed', 'failed'] as const)(
|
||||
'names a later %s command over an earlier Diff',
|
||||
(state) => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
diff('edit', 3, 'src/a.ts'),
|
||||
command('test', 4, 'pnpm test', state)
|
||||
])
|
||||
expect(summary).toMatchObject({ toolName: 'shell', toolInput: 'pnpm test' })
|
||||
}
|
||||
)
|
||||
|
||||
it('names a multi-file edit by its file count, as the chat does', () => {
|
||||
const summary = projectStructuredAgentSessionStatusSummary([
|
||||
ask,
|
||||
running,
|
||||
rg,
|
||||
diff('edit', 4, '3 files')
|
||||
])
|
||||
expect(summary).toMatchObject({ toolName: 'Diff', toolInput: '3 files' })
|
||||
})
|
||||
|
||||
it("never names a subagent's Diff", () => {
|
||||
const childEdit = child(diff('child-edit', 4, 'src/child.ts'))
|
||||
expect(
|
||||
projectStructuredAgentSessionStatusSummary([ask, running, rg, childEdit]).toolName
|
||||
).toBe('shell')
|
||||
expect(statusStructuredAgentSessionToolCall([ask, running, childEdit])).toBeNull()
|
||||
})
|
||||
|
||||
it("names nothing from an ended turn's Diff", () => {
|
||||
const ended = item('running', 2, {
|
||||
kind: 'status',
|
||||
text: 'Done',
|
||||
turnLifecycle: { turnId: 'turn-1', state: 'completed' }
|
||||
})
|
||||
const items = [ask, ended, diff('edit', 3, 'src/a.ts'), followUp]
|
||||
expect(statusStructuredAgentSessionToolCall(items)).toBeNull()
|
||||
expect(
|
||||
projectStructuredAgentSessionStatusSummary(items, [submission('follow-up')]).toolName
|
||||
).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
// The row names the chat's own block, so a tool the chat draws can never read differently here.
|
||||
it('names, at every point in a turn, the block the chat draws for that tool', () => {
|
||||
const timeline: AgentJournalRenderItem[] = [
|
||||
ask,
|
||||
running,
|
||||
item('rg', 3, {
|
||||
kind: 'tool-call',
|
||||
name: 'shell',
|
||||
input: { command: 'rg foo' },
|
||||
callId: 'rg',
|
||||
exitCode: 0,
|
||||
state: 'completed'
|
||||
}),
|
||||
child(item('child-edit', 4, { kind: 'diff', path: 'src/child.ts', patch })),
|
||||
item('edit', 5, { kind: 'diff', path: 'src/a.ts', patch }),
|
||||
item('think', 6, {
|
||||
kind: 'message',
|
||||
role: 'reasoning',
|
||||
blocks: [{ type: 'text', text: 'next' }]
|
||||
}),
|
||||
item('follow-up', 7, {
|
||||
kind: 'message',
|
||||
role: 'user',
|
||||
blocks: [{ type: 'text', text: 'also check the tests' }]
|
||||
}),
|
||||
item('test', 8, {
|
||||
kind: 'tool-call',
|
||||
name: 'shell',
|
||||
input: { command: 'pnpm test' },
|
||||
state: 'running'
|
||||
}),
|
||||
child(
|
||||
item('child-grep', 9, {
|
||||
kind: 'tool-call',
|
||||
name: 'Grep',
|
||||
input: { pattern: 'x' },
|
||||
state: 'running'
|
||||
})
|
||||
),
|
||||
item('multi', 10, { kind: 'diff', path: '2 files', patch })
|
||||
]
|
||||
for (let end = 2; end <= timeline.length; end += 1) {
|
||||
const items = timeline.slice(0, end)
|
||||
const chatCalls = projectStructuredItemsToNativeChat(
|
||||
items.slice(2).filter((entry) => entry.agentId === undefined)
|
||||
)
|
||||
.flatMap((message) => message.blocks)
|
||||
.filter((block) => block.type === 'tool-call')
|
||||
const expected =
|
||||
chatCalls.findLast((block) => block.state === 'running') ?? chatCalls.at(-1) ?? null
|
||||
expect(statusStructuredAgentSessionToolCall(items)).toEqual(expected)
|
||||
const summary = projectStructuredAgentSessionStatusSummary(items)
|
||||
expect(summary.toolName).toBe(expected?.name)
|
||||
expect(summary.toolInput).toBe(expected ? describeToolInput(expected.input) : undefined)
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,46 @@
|
||||
// The one definition of a structured journal's tool actions: which rows the chat draws as a
|
||||
// tool call, and the block it draws. The status line reads the same definition, so the sidebar
|
||||
// names a tool exactly as the chat draws it.
|
||||
|
||||
import type {
|
||||
AgentJournalDiffItem,
|
||||
AgentJournalItemBody,
|
||||
AgentJournalToolCallItem
|
||||
} from './agent-session-journal-types'
|
||||
import type { NativeChatToolCallBlock } from './native-chat-types'
|
||||
|
||||
export type StructuredAgentSessionToolAction = AgentJournalToolCallItem | AgentJournalDiffItem
|
||||
|
||||
// Codex rewrites an edit's `apply_patch` call into a diff once its changes exist.
|
||||
export function isStructuredAgentSessionToolAction(
|
||||
body: AgentJournalItemBody | undefined
|
||||
): body is StructuredAgentSessionToolAction {
|
||||
return body?.kind === 'tool-call' || body?.kind === 'diff'
|
||||
}
|
||||
|
||||
/** A diff carries no lifecycle, so it reads as settled; otherwise a finished edit would stay the
|
||||
* running call for the rest of its turn. */
|
||||
export function isRunningStructuredAgentSessionToolAction(
|
||||
action: StructuredAgentSessionToolAction
|
||||
): boolean {
|
||||
return action.kind === 'tool-call' && action.state === 'running'
|
||||
}
|
||||
|
||||
export function structuredAgentSessionToolCallBlock(
|
||||
action: StructuredAgentSessionToolAction
|
||||
): NativeChatToolCallBlock {
|
||||
if (action.kind === 'diff') {
|
||||
return { type: 'tool-call', name: 'Diff', input: { path: action.path } }
|
||||
}
|
||||
return {
|
||||
type: 'tool-call',
|
||||
name: action.name,
|
||||
input: action.input,
|
||||
state: action.state,
|
||||
...(action.callId !== undefined ? { callId: action.callId } : {}),
|
||||
...(action.mcpIdentity !== undefined ? { mcpIdentity: action.mcpIdentity } : {}),
|
||||
...(action.exitCode !== undefined ? { exitCode: action.exitCode } : {}),
|
||||
...(action.durationMs !== undefined ? { durationMs: action.durationMs } : {}),
|
||||
...(action.webSearchResults !== undefined ? { webSearchResults: action.webSearchResults } : {})
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user