feat: group consecutive tool calls in the AI chat (#11329)

* feat: group consecutive flow edits into one collapsible chat row

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* feat: group lookups, name flow steps and fade tool label changes

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: count rejected draft saves as failed in tool groups

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: address review nits on chat tool grouping

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: hold a tool group's live step line like its header

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: hold a tool group's live line in the same value as its header

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: keep main's new tool cards and held calls out of tool groups

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* feat: group app edits and mark single-server MCP groups with the server icon

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: tell MCP servers apart by full path in tool group headers

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: keep a streaming edit in its group and stop rekeying unsettled labels

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: word a queued tool group as settled until a call starts

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: give queued tool groups their own wording and leave unnamed calls ungrouped

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: keep a tool group in progress between two of its calls

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: treat an edit without a path as unknown unless it is a flow-mode tool

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: let flow mode's pathless edits group again, keep unnamed app edits apart

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: only flow mode's pathless tools default to the open flow

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: key tool rows by their call so a loaded chat never shows a held label

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Co-authored-by: Ruben Fiszel <ruben@windmill.dev>
This commit is contained in:
Guilhem
2026-09-30 14:05:12 +02:00
committed by GitHub
co-authored by Claude Opus 5.5 Ruben Fiszel
parent 083db5e388
commit 9613549abc
12 changed files with 891 additions and 99 deletions
@@ -50,6 +50,8 @@
import { getChatViewHost } from './chatViewHost'
import { getAiChatManager } from './aiChatManagerContext'
import ChatTypingIndicator from './ChatTypingIndicator.svelte'
import ToolGroupDisplay from './ToolGroupDisplay.svelte'
import { chatItemKey, groupToolRuns } from './toolGroups'
import AIChatInput from './AIChatInput.svelte'
import AttachedFilesBar from './files/AttachedFilesBar.svelte'
import QueuedMessageChip from './QueuedMessageChip.svelte'
@@ -237,6 +239,7 @@
const mcpMenu = new McpMenu(aiChatManager, () => assistantSettings?.open('mcp'))
let plusMenuOpen = $state(false)
let editingMessageIndex = $state<number | null>(null)
const chatItems = $derived(groupToolRuns(messages))
// Escape stops the generation when focus is on the chat (or parked on
// body), but stays with other widgets (e.g. the session's Monaco editor).
@@ -872,15 +875,45 @@ the panel, or the Escape-to-stop focus check would wrongly reject them. -->
onscroll={onScroll}
>
<div class="{columnClass} flex flex-col pb-2" bind:clientHeight={height}>
{#each messages as message, messageIndex (messageIndex)}
{#snippet messageRow(
message: DisplayMessage,
messageIndex: number,
isLast: boolean,
inGroup: boolean
)}
<AIChatMessage
{message}
{messageIndex}
{availableContext}
bind:editingMessageIndex
isLast={messageIndex === messages.length - 1}
{isLast}
{inGroup}
showAnswerActions={showsAnswerActions[messageIndex]}
/>
{/snippet}
{#snippet groupRow(message: DisplayMessage, messageIndex: number)}
{@render messageRow(message, messageIndex, false, true)}
{/snippet}
{#each chatItems as item (chatItemKey(item))}
{#if item.kind === 'group'}
<div
class={twMerge(
'mb-1 min-w-0 text-sm px-2 text-primary',
item.entries.at(-1)?.index === messages.length - 1 && '!mb-12'
)}
>
<div class="px-[1px]">
<ToolGroupDisplay group={item} entry={groupRow} />
</div>
</div>
{:else}
{@render messageRow(
item.message,
item.index,
item.index === messages.length - 1,
false
)}
{/if}
{/each}
{#if freeTierExhausted}
{@render freeTierExhaustedBanner()}
@@ -47,6 +47,8 @@
editingMessageIndex: number | null
isLast?: boolean
showAnswerActions?: boolean
/** Rendered inside a tool group, which owns the horizontal padding and the preview chip. */
inGroup?: boolean
}
let {
@@ -55,7 +57,8 @@
availableContext,
editingMessageIndex = $bindable(null),
isLast = false,
showAnswerActions = true
showAnswerActions = true,
inGroup = false
}: Props = $props()
// The edit box edits a copy of THIS message's original context, not the live
@@ -146,7 +149,8 @@
class={twMerge(
'text-sm px-2',
message.role === 'user' && 'py-1',
message.role === 'tool' && 'text-primary'
message.role === 'tool' && 'text-primary',
inGroup && 'px-0'
)}
>
{#if message.role === 'assistant'}
@@ -159,7 +163,10 @@
>
{:else if message.role === 'tool'}
<div class="px-[1px]"
><ToolExecutionDisplay message={message as ToolDisplayMessage} /></div
><ToolExecutionDisplay
message={message as ToolDisplayMessage}
hidePreviewChip={inGroup}
/></div
>
{:else}
{#if message.role === 'user' && message.images && message.images.length > 0}
@@ -1,8 +1,9 @@
<script lang="ts">
import { ChevronRight } from 'lucide-svelte'
import { twMerge } from 'tailwind-merge'
import { slide } from 'svelte/transition'
import { fade, slide } from 'svelte/transition'
import type { Snippet } from 'svelte'
import { HeldValue, LABEL_MIN_MS } from './heldValue.svelte'
interface Props {
label: string
@@ -18,6 +19,13 @@
toggleable?: boolean
// Sweeps a highlight across the label while the row is in progress.
shimmer?: boolean
/** For a label rewritten at each stage of a running call, sometimes several times a
* second: each label (with its shimmer) stays up at least 700 ms, one replaced sooner is
* skipped, and the new one fades in. */
settleLabel?: boolean
/** The step in progress, shown under the header while collapsed. Held with the label, so
* the two change together. */
liveLine?: string
// Ahead of the label, inside the toggle button: a status that reads as part of the
// row rather than as another control, leaving the chevron next to the label it opens.
headerLeft?: Snippet
@@ -39,6 +47,8 @@
onToggle,
toggleable = true,
shimmer = false,
settleLabel = false,
liveLine,
headerLeft,
headerRight,
belowHeader,
@@ -48,6 +58,17 @@
labelClass,
contentClass
}: Props = $props()
// The shimmer and live line are held with the label, in one value: held apart, a call settling
// right after its last status would show the running label, shimmer or line beside the
// settled others.
const held = new HeldValue(
() => ({ prefix: labelPrefix, label, shimmer, liveLine }),
() => (settleLabel ? LABEL_MIN_MS : 0)
)
const shown = $derived(
settleLabel ? held.current : { prefix: labelPrefix, label, shimmer, liveLine }
)
</script>
<div class={twMerge('font-mono text-xs', className)}>
@@ -59,8 +80,8 @@
highlight && 'text-emphasis'
)}
>
{#if labelPrefix}<span class="font-normal text-secondary">{labelPrefix}</span
>&nbsp;{/if}{label}
{#if shown.prefix}<span class="font-normal text-secondary">{shown.prefix}</span
>&nbsp;{/if}{shown.label}
</span>
{/snippet}
@@ -75,16 +96,27 @@
aria-expanded={toggleable ? expanded : undefined}
>
{@render headerLeft?.()}
{#if shimmer}
<span class="shimmer inline-flex items-center min-w-0">
{@render labelText(false)}
<span class="shimmer-band inline-flex items-center min-w-0" aria-hidden="true">
{@render labelText(true)}
</span>
<!-- A settled label fades in. The key sits outside the shimmer branch because a label
change often comes with a shimmer change, and a transition inside a branch being
swapped out does not play. In only: an outgoing copy would widen the row. Unsettled
labels (an agent trace's ticking duration) keep one key and update in place. -->
{#key settleLabel ? `${shown.prefix ?? ''}\n${shown.label}` : ''}
<span
class="inline-flex items-center min-w-0"
in:fade={{ duration: settleLabel ? 400 : 0 }}
>
{#if shown.shimmer}
<span class="shimmer inline-flex items-center min-w-0">
{@render labelText(false)}
<span class="shimmer-band inline-flex items-center min-w-0" aria-hidden="true">
{@render labelText(true)}
</span>
</span>
{:else}
{@render labelText(false)}
{/if}
</span>
{:else}
{@render labelText(false)}
{/if}
{/key}
{#if toggleable}
<ChevronRight
class={twMerge(
@@ -105,6 +137,9 @@
{@render headerButton()}
{/if}
{#if shown.liveLine && !expanded}
<div class="pl-3 text-2xs text-tertiary truncate">{shown.liveLine}</div>
{/if}
{@render belowHeader?.()}
{#if expanded && children}
@@ -51,9 +51,10 @@
interface Props {
message: ToolDisplayMessage
hidePreviewChip?: boolean
}
let { message }: Props = $props()
let { message, hidePreviewChip = false }: Props = $props()
// Recorded by the call itself, from the connected-server list rather than from the
// model's arguments — which is what lets a reloaded transcript still resolve it, and
@@ -142,7 +143,11 @@
// shown once the tool settled, never while loading/erroring/awaiting confirmation.
const showPreviewChip = $derived(
Boolean(
message.previewCard && !message.isLoading && !message.error && !message.needsConfirmation
!hidePreviewChip &&
message.previewCard &&
!message.isLoading &&
!message.error &&
!message.needsConfirmation
)
)
@@ -317,6 +322,7 @@
onToggle={() => (isExpanded = !isExpanded)}
toggleable={detailsAvailable || message.isStreamingArguments === true}
shimmer={isRunning}
settleLabel
class={message.isQueued && !message.error
? 'opacity-60 hover:opacity-100 transition-opacity'
: ''}
@@ -0,0 +1,87 @@
<script lang="ts">
import type { Snippet } from 'svelte'
import ChatCollapsibleCard from './ChatCollapsibleCard.svelte'
import ToolPreviewCard from './ToolPreviewCard.svelte'
import type { DisplayMessage, ToolDisplayMessage } from './shared'
import { callFailed, groupHeader, type ToolGroup } from './toolGroups'
import McpServerIcon from '$lib/components/mcp/McpServerIcon.svelte'
import { resolveMcpServerMark } from '$lib/components/mcp/serverMark'
interface Props {
group: ToolGroup
entry: Snippet<[DisplayMessage, number]>
}
let { group, entry }: Props = $props()
let expanded = $state(false)
const calls = $derived(
group.entries.map((e) => e.message).filter((m): m is ToolDisplayMessage => m.role === 'tool')
)
// Same states as a single row: a group whose calls are all still waiting their turn is faded
// like a queued row. Once one call has run, the group is in progress until every call has,
// including the moments between two calls when none is executing.
const executing = $derived(calls.some((m) => m.isLoading || m.isStreamingArguments))
const waiting = $derived(calls.some((m) => m.isQueued))
const queued = $derived(!executing && waiting && calls.every((m) => m.isQueued))
const running = $derived(executing || (waiting && !queued))
const header = $derived(groupHeader(group, running ? 'running' : queued ? 'queued' : 'settled'))
const failedCount = $derived(calls.filter(callFailed).length)
// Every edit of one item carries the same chip, so the group shows it once.
const previewCard = $derived(calls.findLast((m) => m.previewCard && !m.error)?.previewCard)
// An explore header has a prefix only when every call is an MCP read on one server; that
// group is marked with the server's icon, as a single MCP row is.
const mcpServer = $derived(
group.groupKind === 'explore' && header.prefix
? calls.find((m) => m.mcpServer)?.mcpServer
: undefined
)
// What the group is doing right now, so a collapsed run still names its current step.
const liveLine = $derived(
running
? (
calls.findLast((m) => m.isLoading || m.isStreamingArguments) ??
calls.find((m) => m.isQueued)
)?.content
: undefined
)
</script>
{#snippet serverMark()}
{#if mcpServer}
{#await resolveMcpServerMark(mcpServer.workspace, mcpServer.path) then mark}
<McpServerIcon icon={mark.icon} size={14} />
{/await}
{/if}
{/snippet}
{#snippet status()}
<div class="flex items-center gap-2 shrink-0">
{#if failedCount > 0}
<span class="font-main text-2xs text-red-600 dark:text-red-400">{failedCount} failed</span>
{/if}
{#if previewCard}
<ToolPreviewCard card={previewCard} />
{/if}
</div>
{/snippet}
<ChatCollapsibleCard
labelPrefix={header.prefix}
label={header.label}
{expanded}
onToggle={() => (expanded = !expanded)}
shimmer={running}
settleLabel
{liveLine}
class={queued ? 'opacity-60 hover:opacity-100 transition-opacity' : ''}
labelClass="truncate"
headerRight={failedCount > 0 || previewCard ? status : undefined}
headerLeft={mcpServer ? serverMark : undefined}
contentClass="border-0 border-l rounded-none bg-transparent p-0 pl-2 ml-1.5 mt-0.5"
>
{#each group.entries as { message, index } (index)}
{@render entry(message, index)}
{/each}
</ChatCollapsibleCard>
@@ -0,0 +1,4 @@
/** `result` of a draft write that did not land. The write reports it through `result` alone, not
* `error`, so a tool group reads these to count the call as failed. */
export const DRAFT_CONFLICT_RESULT = 'Conflict'
export const DRAFT_SAVE_FAILED_RESULT = 'Save failed'
@@ -250,6 +250,8 @@ import {
saveGlobalAppDraft,
type DraftPersistResult
} from './userDraftAdapter'
import { findModuleInFlow } from '$lib/components/flows/flowTree'
import { DRAFT_CONFLICT_RESULT, DRAFT_SAVE_FAILED_RESULT } from '../draftWriteResults'
import {
computeDiffParts,
expireWorkspaceDiffList,
@@ -5125,7 +5127,7 @@ function draftWriteFailure(result: DraftPersistResult, ctx: WriteDraftCtx): stri
if (result.status === 'conflict') {
ctx.toolCallbacks.setToolStatus(ctx.toolId, {
content: `Draft ${stored.type} "${stored.path}" changed externally`,
result: `Conflict`
result: DRAFT_CONFLICT_RESULT
})
return JSON.stringify(
{
@@ -5140,7 +5142,7 @@ function draftWriteFailure(result: DraftPersistResult, ctx: WriteDraftCtx): stri
if (result.status === 'error') {
ctx.toolCallbacks.setToolStatus(ctx.toolId, {
content: `Failed to save ${stored.type} "${stored.path}"`,
result: `Save failed`
result: DRAFT_SAVE_FAILED_RESULT
})
return JSON.stringify(
{
@@ -5710,7 +5712,7 @@ async function readFlowModuleCode(
)
}
toolCallbacks.setToolStatus(toolId, {
content: `Read inline script for "${args.module_id}"`
content: `Read code of step ${flowStepName(base.flow.value, args.module_id)}`
})
return content
}
@@ -5733,7 +5735,7 @@ async function setFlowModuleCode(
}
session.set(args.module_id, args.code)
const newFlowValue = applyEditableFlowJsonToFlow(base.flow.value, editable, session)
return writeFlowDraft(
const result = await writeFlowDraft(
{
path: args.path,
summary: base.summary,
@@ -5741,6 +5743,18 @@ async function setFlowModuleCode(
},
ctx
)
// Several code edits of one flow read as identical rows under the generic flow label.
if (JSON.parse(result).success) {
toolCallbacks.setToolStatus(toolId, {
content: `Updated code of step ${flowStepName(base.flow.value, args.module_id)}`
})
}
return result
}
function flowStepName(flow: FlowValue, moduleId: string): string {
const summary = findModuleInFlow(flow, moduleId)?.summary
return summary ? `${moduleId} "${summary}"` : moduleId
}
function normalizeTestRunArgs(args: Record<string, any> | null | undefined): Record<string, any> {
@@ -0,0 +1,43 @@
import { watch } from 'runed'
/** How long a tool-call status label stays up before a newer one replaces it. */
export const LABEL_MIN_MS = 700
/**
* A value that, once shown, stays shown for at least `minMs`. A change arriving sooner waits
* for the rest of that time, and only the latest of the changes made meanwhile is shown. The
* first value starts the window too, so a change right after mount is held like any other.
* Must be constructed during component initialisation.
*/
export class HeldValue<T> {
#current = $state() as T
#shownAt = Date.now()
constructor(getter: () => T, minMs: () => number) {
this.#current = getter()
watch(
getter,
(value) => {
const wait = this.#shownAt + minMs() - Date.now()
if (wait <= 0) {
this.#show(value)
return
}
// Re-run on every change, so the cleanup drops the pending value in favour of the
// newer one, and also clears the timer on unmount.
const timer = setTimeout(() => this.#show(value), wait)
return () => clearTimeout(timer)
},
{ lazy: true }
)
}
#show(value: T) {
this.#current = value
this.#shownAt = Date.now()
}
get current(): T {
return this.#current
}
}
@@ -62,6 +62,7 @@ import { scriptLangToEditorLang } from '$lib/scripts'
import { getCurrentModel } from '$lib/aiStore'
import type { editor as meditor } from 'monaco-editor'
import { pendingFolderInstructions, type FolderInstructionsContext } from './folderInstructions'
import type { WebSearchSource } from './webSearchResult'
// Prettify function for code arguments - extracts and formats code from JSON
function prettifyCodeArguments(content: string): string {
@@ -607,11 +608,12 @@ export type RunFormDraft = {
schema: Record<string, any>
}
/** One page hit from a provider-side web search (OpenAI sources carry no title). */
export type WebSearchSource = {
url: string
title?: string
}
export {
isRenderableSourceUrl,
webSearchResultOf,
type WebSearchResult,
type WebSearchSource
} from './webSearchResult'
export type ToolCodeDiff = {
before: string
@@ -620,77 +622,6 @@ export type ToolCodeDiff = {
lang: string
}
/** The result shape any tool returns to have it rendered as a web search card. */
export type WebSearchResult = {
sources: WebSearchSource[]
query?: string
}
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === 'object' && value !== null && !Array.isArray(value)
}
/** A link the card can render: the source list drops everything else, so a result whose
* urls are relative or `javascript:` would show as an empty list in place of its own output. */
export function isRenderableSourceUrl(url: string): boolean {
try {
return ['http:', 'https:'].includes(new URL(url).protocol)
} catch {
return false
}
}
/** An absent optional field reaches JSON as `null` from a Python tool and as nothing from a
* TypeScript one, so both read as absent. */
function isOptionalString(value: unknown): boolean {
return value === null || typeof value === 'string'
}
function isWebSearchSource(value: unknown): value is { url: string; title?: unknown } {
return (
isRecord(value) &&
typeof value.url === 'string' &&
isRenderableSourceUrl(value.url) &&
Object.entries(value).every(
([key, v]) => key === 'url' || (key === 'title' && isOptionalString(v))
)
)
}
/**
* A tool result as a web search, or undefined when it is anything else. The keys must be
* exactly `sources` and an optional `query`: the card renders only urls and titles, so a
* result carrying more would lose it. Accepts the result as its JSON text too.
*/
export function webSearchResultOf(result: unknown): WebSearchResult | undefined {
if (typeof result === 'string') {
try {
result = JSON.parse(result)
} catch {
return undefined
}
}
if (!isRecord(result) || !Array.isArray(result.sources) || result.sources.length === 0) {
return undefined
}
const sources = result.sources
if (
!sources.every(isWebSearchSource) ||
!Object.entries(result).every(
([key, v]) => key === 'sources' || (key === 'query' && isOptionalString(v))
)
) {
return undefined
}
return {
sources: sources.map((source) => ({
url: source.url,
title: typeof source.title === 'string' ? source.title : undefined
})),
query: typeof result.query === 'string' ? result.query : undefined
}
}
export type ToolDisplayMessage = {
role: 'tool'
tool_call_id: string
@@ -0,0 +1,226 @@
import { describe, expect, it } from 'vitest'
import type { DisplayMessage } from './shared'
import { groupHeader, groupToolRuns, type GroupState, type ToolGroup } from './toolGroups'
let nextId = 0
function tool(
toolName: string,
parameters: Record<string, unknown> = {},
extra = {}
): DisplayMessage {
return {
role: 'tool',
tool_call_id: `call_${nextId++}`,
content: toolName,
toolName,
parameters,
...extra
}
}
function assistant(content: string, extra = {}): DisplayMessage {
return { role: 'assistant', content, ...extra } as DisplayMessage
}
function shape(messages: DisplayMessage[]) {
return groupToolRuns(messages).map((item) =>
item.kind === 'group' ? { [item.groupKind]: item.entries.map((e) => e.index) } : item.index
)
}
function header(messages: DisplayMessage[], state: GroupState = 'settled') {
const group = groupToolRuns(messages).find((item) => item.kind === 'group') as ToolGroup
const { prefix, label } = groupHeader(group, state)
return prefix ? `${prefix} ${label}` : label
}
describe('groupToolRuns', () => {
it('folds edits and reads of one flow, with thinking in between, into one group', () => {
const flow = { path: 'f/a/flow' }
expect(
shape([
assistant('Let me edit the flow.'),
tool('read_flow_module_code', flow),
tool('patch_flow_json', flow),
assistant('', { reasoning: 'next step' }),
tool('set_flow_module_code', flow),
assistant('', { reasoning: 'done' }),
assistant('All steps updated.')
])
).toEqual([0, { edit: [1, 2, 3, 4] }, 5, 6])
})
it('splits on a different flow, on visible text, and leaves a lone edit ungrouped', () => {
const a = { path: 'f/a/flow' }
const b = { path: 'f/b/flow' }
expect(
shape([
tool('patch_flow_json', a),
tool('patch_flow_json', a),
tool('patch_flow_json', b),
assistant('Now the other one.'),
tool('patch_flow_json', b)
])
).toEqual([{ edit: [0, 1] }, 2, 3, 4])
})
it('never folds a call waiting for confirmation, and folds reads without an edit as exploring', () => {
const flow = { path: 'f/a/flow' }
expect(
shape([
tool('patch_flow_json', flow),
tool('patch_flow_json', flow, { needsConfirmation: true, isLoading: true }),
tool('read_flow_module_code', flow),
tool('read_flow_module_code', flow)
])
).toEqual([0, 1, { explore: [2, 3] }])
expect(
shape([
tool('patch_flow_json', flow),
tool('patch_flow_json', flow, { declinedByUser: true, error: 'Cancelled by user' }),
tool('patch_flow_json', flow)
])
).toEqual([0, 1, 2])
const sources = JSON.stringify({ sources: [{ url: 'https://example.com', title: 'x' }] })
expect(
shape([
tool('search_docs'),
tool('call_mcp_read_tool', { server: 'u/a/s', tool: 'search' }, { result: sources }),
tool('patch_flow_json', flow, { heldForFolderInstructions: true }),
tool('patch_flow_json', flow)
])
).toEqual([0, 1, 2, 3])
})
it('keeps an edit whose arguments are still streaming in the group it follows', () => {
const flow = { path: 'f/a/flow' }
expect(
shape([
tool('patch_flow_json', flow),
tool('patch_flow_json', '{"path":"f/a/flow","old_str' as never, {
isStreamingArguments: true
}),
tool('set_flow_module_code', '{"pa' as never, { isStreamingArguments: true })
])
).toEqual([{ edit: [0, 1, 2] }])
expect(
shape([
tool('patch_flow_json', flow),
tool('patch_flow_json', '{"path":"f/b/flow","old_str' as never, {
isStreamingArguments: true
})
])
).toEqual([0, 1])
})
it('lets a call with no arguments yet join a group but not start one', () => {
const queued = (toolName: string) =>
tool(toolName, {}, { parameters: undefined, isQueued: true })
expect(shape([queued('delete_app_file'), queued('delete_app_runnable')])).toEqual([0, 1])
expect(shape([tool('patch_app_file', { path: 'f/a/x' }), queued('delete_app_file')])).toEqual([
{ edit: [0, 1] }
])
// Arguments without a path are a call the tool will reject, not an edit of the open
// flow; only flow-mode tools, which never take a path, share the open flow.
expect(
shape([tool('write_app_file', { file_path: '/a' }), tool('patch_app_file', {})])
).toEqual([0, 1])
expect(
shape([
tool('set_module_code', { moduleId: 'a' }),
tool('patch_flow_json', { old_string: 'a', new_string: 'b' }),
tool('set_module_code', { moduleId: 'b' })
])
).toEqual([{ edit: [0, 1, 2] }])
// A global-only flow tool always takes a path, so without one its item is unknown.
expect(shape([tool('write_flow', { summary: 'x' }), tool('set_flow_module_code', {})])).toEqual(
[0, 1]
)
})
it('folds edits and reads of one app, apart from a flow at the same path', () => {
const app = { path: 'f/a/x', file_path: '/src/App.tsx' }
expect(
shape([
tool('read_app_file', app),
tool('patch_app_file', app),
tool('write_app_runnable', { path: 'f/a/x', key: 'fetch' }),
tool('patch_flow_json', { path: 'f/a/x' }),
tool('patch_flow_json', { path: 'f/a/x' })
])
).toEqual([{ edit: [0, 1, 2] }, { edit: [3, 4] }])
})
it('leaves a flow read that prepares an edit to the edit group, not the lookups before it', () => {
const flow = { type: 'flow', path: 'f/a/flow' }
expect(
shape([
tool('search_workspace'),
tool('list_workspace_items'),
tool('read_workspace_item', flow),
tool('patch_flow_json', flow)
])
).toEqual([{ explore: [0, 1] }, { edit: [2, 3] }])
})
it('folds consecutive lookups of any tool, split by a write or a row with its own card', () => {
expect(
shape([
tool('search_workspace'),
assistant('', { reasoning: 'look closer' }),
tool('read_workspace_item', { type: 'script', path: 'f/a/s' }),
tool('get_run', {}, { inspectedRun: { jobId: 'x' } }),
tool('list_runs'),
tool('search_docs'),
tool('write_script', { path: 'f/a/s' }),
tool('search_workspace')
])
).toEqual([{ explore: [0, 1, 2] }, 3, { explore: [4, 5] }, 6, 7])
})
it('names what a group did', () => {
const flow = { path: 'f/a/flow' }
expect(
header([
tool('read_flow_module_code', flow),
tool('patch_flow_json', flow),
tool('patch_flow_json', flow)
])
).toBe('Edited f/a/flow · 2 changes')
// A save rejected as a conflict reports it in `result`, not `error`: not a change.
expect(
header([
tool('patch_flow_json', flow),
tool('patch_flow_json', flow, { result: 'Conflict' }),
tool('set_flow_module_code', flow, { result: 'Save failed' })
])
).toBe('Edited f/a/flow · 1 change')
// Before any call runs, the header says what will happen, not what did.
expect(
header(
[
tool('patch_flow_json', flow, { isQueued: true }),
tool('patch_flow_json', flow, { isQueued: true })
],
'queued'
)
).toBe('Edit f/a/flow · 2 changes')
expect(
header([
tool('search_workspace'),
tool('read_workspace_item'),
tool('search_workspace'),
tool('read_workspace_item'),
tool('list_runs')
])
).toBe('Search workspace 2 times, read 2 workspace item, list runs')
const mcp = (t: string) => tool('call_mcp_read_tool', { server: 'u/admin/github', tool: t })
expect(header([mcp('list_issues'), mcp('get_issue'), mcp('get_issue')])).toBe(
'github list issues, get issue 2 times'
)
expect(header([tool('search_mcp_tools'), mcp('list_issues'), mcp('get_issue')])).toBe(
'github search mcp tools, list issues, get issue'
)
const other = tool('call_mcp_read_tool', { server: 'f/team/github', tool: 'get_issue' })
expect(header([mcp('list_issues'), other])).toBe('List issues, get issue')
})
})
@@ -0,0 +1,330 @@
import type { DisplayMessage, ToolDisplayMessage } from './shared'
import { DRAFT_CONFLICT_RESULT, DRAFT_SAVE_FAILED_RESULT } from './draftWriteResults'
import { webSearchResultOf } from './webSearchResult'
// A tool missing from these lists always renders as its own row, so a new write never gets
// hidden by default.
type EditedItemKind = 'flow' | 'app'
// Tools that fold into an edit group, by the kind of item they edit; the item is the call's
// `path` ('' for flow mode's tools, which edit the open flow). Scripts are left out on purpose: each script edit
// renders its own diff card, and the diff is what the user reads.
const EDIT_TOOLS: Record<string, EditedItemKind> = {
patch_flow_json: 'flow',
set_flow_module_code: 'flow',
write_flow: 'flow',
set_flow_json: 'flow',
set_module_code: 'flow',
set_preprocessor_module: 'flow',
set_failure_module: 'flow',
init_app: 'app',
write_app_file: 'app',
patch_app_file: 'app',
delete_app_file: 'app',
write_app_runnable: 'app',
delete_app_runnable: 'app'
}
// Reads of the item being edited: the model reads a step or file before changing it, and
// splitting the group on every read would leave one group per edit.
const ITEM_READ_TOOLS: Record<string, EditedItemKind> = {
read_flow_module_code: 'flow',
inspect_inline_script: 'flow',
get_lint_errors: 'flow',
read_app_file: 'app',
search_app: 'app'
}
// The grouped tools flow mode (flow/core.ts) offers without a `path`: they act on the flow open
// in the editor. patch_flow_json shares its name with the global tool, which takes a path.
const PATHLESS_FLOW_MODE_TOOLS = new Set([
'set_flow_json',
'patch_flow_json',
'set_module_code',
'set_preprocessor_module',
'set_failure_module',
'inspect_inline_script',
'get_lint_errors'
])
// Calls that only look things up. Not derived from `planModeSafe`, which also admits test
// runs and plan-document writes.
const READ_TOOLS = new Set([
...Object.keys(ITEM_READ_TOOLS),
'list_workspace_items',
'read_workspace_item',
'search_workspace',
'get_runnable_details',
'search_hub_scripts',
'search_resource_types',
'resource_type',
'search_docs',
'read_docs_page',
'get_db_schema',
'search_npm_packages',
'get_instructions',
'get_instructions_for_code_generation',
'read_skill',
'get_trigger_schema',
'get_schedule_schema',
'list_runs',
'list_workers',
'list_app_runs',
'get_app_runtime_logs',
'get_preview_status',
'get_current_page_name',
'search_dom',
'read_dom',
'read_file',
'search_files',
'list_data_metrics',
'list_ducklakes',
'get_pipeline_graph',
'read_pipeline_node',
'list_artifacts',
'read_artifact',
'list_artifact_versions',
'search_mcp_tools',
'call_mcp_read_tool'
])
export type ToolGroup = {
kind: 'group'
/** 'edit': edits of one flow or app. 'explore': consecutive lookups of anything. */
groupKind: 'edit' | 'explore'
/** First call's id, so the group keeps its identity (and expand state) as it grows. */
key: string
/** Edit groups: the item's path, or '' for the flow open in the editor. */
target: string
entries: { message: DisplayMessage; index: number }[]
}
export type ChatItem = { kind: 'message'; message: DisplayMessage; index: number } | ToolGroup
/** A tool row is keyed by its call, not its position: loading another chat puts a different
* call at the same index, and a reused row would carry the old call's held label over. */
export function chatItemKey(item: ChatItem): string {
if (item.kind === 'group') return `g:${item.key}`
if (item.message.role === 'tool') return `t:${item.message.tool_call_id}`
return `m:${item.index}`
}
function groupableCall(message: DisplayMessage): ToolDisplayMessage | undefined {
if (message.role !== 'tool' || !message.toolName) return undefined
// A row waiting on the user, or refused by plan mode, is a decision the user must see; a
// row with its own card (run, question, diff, image, sources) is the content itself; a call
// held for folder instructions never ran, so it is not a change or a lookup.
if (message.needsConfirmation || message.blockedByPlanMode) return undefined
if (message.heldForFolderInstructions) return undefined
if (message.runForm || message.inspectedRun || message.userQuestion) return undefined
if (message.codeDiff || message.imageUrl || message.webSearchSources) return undefined
if (!message.error && webSearchResultOf(message.result)) return undefined
// The user's own refusal is a decision, not a failure to fold away.
if (message.declinedByUser) return undefined
return message
}
/** `path` undefined: the call's arguments have not named its item yet (still streaming, or
* not on the row yet). */
type Membership = { kind: EditedItemKind; path: string | undefined; edit: boolean }
// While arguments stream, `parameters` is the partial JSON text. The path usually lands well
// before the large code or content field, so read it out once its closing quote has arrived.
function streamedPath(text: string): string | undefined {
const match = /"path"\s*:\s*"((?:[^"\\]|\\.)*)"/.exec(text)
if (!match) return undefined
try {
return JSON.parse(`"${match[1]}"`)
} catch {
return undefined
}
}
function itemMembership(message: DisplayMessage): Membership | undefined {
const call = groupableCall(message)
if (!call) return undefined
const streaming = typeof call.parameters === 'string'
const params = streaming ? {} : (call.parameters ?? {})
const tool = call.toolName!
// No `parameters` at all: a call queued before its arguments reached the row (tools that do
// not stream them), so its item is not known yet. Arguments without a path mean the open
// flow only for a tool flow mode offers without one; any other tool requires a path, so
// its item is unknown.
const path = streaming
? streamedPath(call.parameters)
: call.parameters === undefined
? undefined
: typeof params.path === 'string'
? params.path
: PATHLESS_FLOW_MODE_TOOLS.has(tool)
? ''
: undefined
if (Object.hasOwn(EDIT_TOOLS, tool)) return { kind: EDIT_TOOLS[tool], path, edit: true }
if (Object.hasOwn(ITEM_READ_TOOLS, tool)) {
return { kind: ITEM_READ_TOOLS[tool], path, edit: false }
}
if (tool === 'read_workspace_item' && (params.type === 'flow' || params.type === 'app') && path) {
return { kind: params.type, path, edit: false }
}
return undefined
}
// A flow and an app can share a path, so the kind is part of what makes two calls one item. A
// call whose path has not arrived yet stays with the group it follows, or the edit in progress
// would leave its own group and the header would read as settled.
function sameItem(a: Membership | undefined, b: Membership): boolean {
return a !== undefined && a.kind === b.kind && (a.path === undefined || a.path === b.path)
}
function isReadCall(message: DisplayMessage): boolean {
const call = groupableCall(message)
return call !== undefined && READ_TOOLS.has(call.toolName!)
}
// Thinking between two calls stays inside the group; visible text ends it. The live
// streaming message is never absorbed, or the reasoning in progress would be hidden.
function isSilentAssistant(message: DisplayMessage): boolean {
return message.role === 'assistant' && !message.streaming && message.content.trim() === ''
}
/** Index of the last call in the run starting at `start`, crossing silent assistant
* messages; trailing silent messages stay outside. */
function runEnd(
messages: DisplayMessage[],
start: number,
accepts: (m: DisplayMessage, index: number) => boolean
) {
let end = start
for (let j = start + 1; j < messages.length; j++) {
if (accepts(messages[j], j)) end = j
else if (!isSilentAssistant(messages[j])) break
}
return end
}
function toolGroup(
messages: DisplayMessage[],
start: number,
end: number,
groupKind: ToolGroup['groupKind'],
target: string
): ToolGroup | undefined {
const run = messages.slice(start, end + 1)
const calls = run.filter((m) => m.role === 'tool')
if (calls.length < 2) return undefined
if (groupKind === 'edit' && !calls.some((m) => itemMembership(m)?.edit)) return undefined
return {
kind: 'group',
groupKind,
key: (messages[start] as ToolDisplayMessage).tool_call_id,
target,
entries: run.map((message, k) => ({ message, index: start + k }))
}
}
function editGroupAt(messages: DisplayMessage[], start: number): ToolGroup | undefined {
const item = itemMembership(messages[start])
// A call whose item is not known yet may join a group but not start one: two such calls
// could be edits of different items.
if (!item || item.path === undefined) return undefined
const end = runEnd(messages, start, (m) => sameItem(itemMembership(m), item))
return toolGroup(messages, start, end, 'edit', item.path)
}
export function groupToolRuns(messages: DisplayMessage[]): ChatItem[] {
const items: ChatItem[] = []
let i = 0
while (i < messages.length) {
// An edit group wins over an explore group: its reads belong to the edits they prepare,
// so an explore run also stops before a read that starts one.
const explores = (m: DisplayMessage, j: number) => isReadCall(m) && !editGroupAt(messages, j)
const group =
editGroupAt(messages, i) ??
(isReadCall(messages[i])
? toolGroup(messages, i, runEnd(messages, i, explores), 'explore', '')
: undefined)
if (group) {
items.push(group)
i = group.entries.at(-1)!.index + 1
} else {
items.push({ kind: 'message', message: messages[i], index: i })
i++
}
}
return items
}
// The server's full resource path: two servers can share a last segment (u/alice/github,
// f/team/github), so only the display shortens it.
function mcpServerPath(call: ToolDisplayMessage): string | undefined {
if (call.toolName !== 'call_mcp_read_tool') return undefined
const server = call.mcpServer?.path ?? call.parameters?.server
return typeof server === 'string' ? server : undefined
}
function callName(call: ToolDisplayMessage): string {
const mcpTool = call.parameters?.tool
const name =
call.toolName === 'call_mcp_read_tool' && typeof mcpTool === 'string'
? mcpTool
: (call.toolName ?? '')
return name.replaceAll('_', ' ')
}
// A collapsed group would otherwise hide a draft save that did not land.
const FAILED_SAVE_RESULTS = new Set([DRAFT_CONFLICT_RESULT, DRAFT_SAVE_FAILED_RESULT])
export function callFailed(call: ToolDisplayMessage): boolean {
return (
call.error !== undefined ||
(typeof call.result === 'string' && FAILED_SAVE_RESULTS.has(call.result))
)
}
function plural(n: number, word: string): string {
return `${n} ${word}${n === 1 ? '' : 's'}`
}
// Each read reads one thing, so the count goes on the object ("read 3 workspace item"). On
// other verbs it would miscount what came back ("list 2 runs" after listing twice).
const COUNTED_VERBS = new Set(['read', 'inspect'])
function repeated(name: string, n: number): string {
if (n === 1) return name
const [verb, ...rest] = name.split(' ')
return COUNTED_VERBS.has(verb) && rest.length > 0
? `${verb} ${n} ${rest.join(' ')}`
: `${name} ${n} times`
}
/** `queued`: every call is still waiting its turn. Every call in a turn is queued before the
* first one runs, so this state shows on each edit run; like a single row's queued label, it
* says what will happen rather than what did. */
export type GroupState = 'queued' | 'running' | 'settled'
/** The group's header, e.g. `Edited` + `f/a/flow · 3 changes`,
* `search workspace 2 times, read 2 workspace item`, `github` + `list issues, get issue 2 times`. */
export function groupHeader(
group: ToolGroup,
state: GroupState
): { prefix: string; label: string } {
const calls = group.entries
.map((e) => e.message)
.filter((m): m is ToolDisplayMessage => m.role === 'tool')
if (group.groupKind === 'edit') {
const edits = calls.filter(
(m) => Object.hasOwn(EDIT_TOOLS, m.toolName ?? '') && !callFailed(m)
).length
return {
prefix: { queued: 'Edit', running: 'Editing', settled: 'Edited' }[state],
label: `${group.target || 'the flow'} · ${plural(edits, 'change')}`
}
}
const counts = new Map<string, number>()
for (const call of calls) counts.set(callName(call), (counts.get(callName(call)) ?? 0) + 1)
const list = [...counts].map(([name, n]) => repeated(name, n)).join(', ')
// A tool search spans every server and usually comes right before the calls it found, so it
// does not stop the group from reading as one server's.
const servers = new Set(
calls.filter((call) => call.toolName !== 'search_mcp_tools').map(mcpServerPath)
)
const [server] = servers
if (servers.size === 1 && server) return { prefix: server.split('/').at(-1)!, label: list }
return { prefix: '', label: list.charAt(0).toUpperCase() + list.slice(1) }
}
@@ -0,0 +1,76 @@
/** One page hit from a provider-side web search (OpenAI sources carry no title). */
export type WebSearchSource = {
url: string
title?: string
}
/** The result shape any tool returns to have it rendered as a web search card. */
export type WebSearchResult = {
sources: WebSearchSource[]
query?: string
}
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === 'object' && value !== null && !Array.isArray(value)
}
/** A link the card can render: the source list drops everything else, so a result whose
* urls are relative or `javascript:` would show as an empty list in place of its own output. */
export function isRenderableSourceUrl(url: string): boolean {
try {
return ['http:', 'https:'].includes(new URL(url).protocol)
} catch {
return false
}
}
/** An absent optional field reaches JSON as `null` from a Python tool and as nothing from a
* TypeScript one, so both read as absent. */
function isOptionalString(value: unknown): boolean {
return value === null || typeof value === 'string'
}
function isWebSearchSource(value: unknown): value is { url: string; title?: unknown } {
return (
isRecord(value) &&
typeof value.url === 'string' &&
isRenderableSourceUrl(value.url) &&
Object.entries(value).every(
([key, v]) => key === 'url' || (key === 'title' && isOptionalString(v))
)
)
}
/**
* A tool result as a web search, or undefined when it is anything else. The keys must be
* exactly `sources` and an optional `query`: the card renders only urls and titles, so a
* result carrying more would lose it. Accepts the result as its JSON text too.
*/
export function webSearchResultOf(result: unknown): WebSearchResult | undefined {
if (typeof result === 'string') {
try {
result = JSON.parse(result)
} catch {
return undefined
}
}
if (!isRecord(result) || !Array.isArray(result.sources) || result.sources.length === 0) {
return undefined
}
const sources = result.sources
if (
!sources.every(isWebSearchSource) ||
!Object.entries(result).every(
([key, v]) => key === 'sources' || (key === 'query' && isOptionalString(v))
)
) {
return undefined
}
return {
sources: sources.map((source) => ({
url: source.url,
title: typeof source.title === 'string' ? source.title : undefined
})),
query: typeof result.query === 'string' ? result.query : undefined
}
}