From 7faa9f7cd3bda68189646b1e9def6cc8aae3b1cb Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:19:24 -0700 Subject: [PATCH] fix(native-chat): reasoning rows with a real open/finished state, and readable Claude thinking (#19221) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(native-chat): render structured reasoning as collapsible messages * fix(native-chat): align expanded reasoning with summary * fix(native-chat): place reasoning chevron after summary * De-emphasize reasoning headlines with observed timing labels * Exclude later turn work from observed reasoning duration * fix(native-chat): give reasoning rows a host-owned open/closed lifecycle A reasoning row now says whether its block is still streaming (`state`) and, when the host saw it end, when (`completedAt`). The row's own start is its first-write time, so "Thought for N s" is measured on the execution host instead of by whichever window happened to be watching. Rows open with their first non-empty text and close on every path that ends them: the block's final frame, a new message in the same stream, every Claude turn end through one hook on the open turn, Codex item/completed, turn and session settlement, active-item eviction, and both host sweeps for a dead generation. Closing writes are lifecycle writes so backpressure cannot leave a row open. Rows without the field (older hosts, older journals) never read as live. The reasoning row's headline reads the row's lifecycle and its own turn's liveness: Thinking while open in a running turn, then Thought for N s, Thought when no span was seen, Reasoning when the host kept no lifecycle. * feat(native-chat): ask Claude for readable thinking summaries Under Orca's launch the Claude CLI streams thinking blocks with empty text, so no reasoning row ever had anything to show. Pass only `--thinking-display summarized`: it fills thinking blocks with the API's summaries without turning thinking on, so a user who disabled thinking keeps it off. Orca runs the user's own binary, and a CLI older than the flag exits on it before the session starts. The launch probes the binary it is about to run, overlapping the rest of launch resolution, and passes the flag only when that probe has already answered with 2.1.94 or newer. A slow or failed probe never delays a launch and is never remembered; a successful one is kept per binary until the binary changes. * test(native-chat): type the superseding send in the reasoning lifecycle test * fix(native-chat): stop reading an ended reasoning row as thinking The spinner line infers "Thinking" from the newest root row being a reasoning message. With summaries on, a closed reasoning row stays the newest row while Claude streams a tool's input, so the line read Thinking for the whole Write. A reasoning row now counts only while its own state is running; a row from a host that keeps no state reads as before. * fix(native-chat): measure reasoning from the block's start, not its first text A row opens with its first summary text, which trails the thinking block's start by seconds, and the journal stamps a row with its queued append time. So "Thought for N s" read 1 s for blocks that ran 4.65 s and 12.95 s. The block registry now records the host time at each block's content_block_start (else its first delta), and every write of that block's row carries it as the row's observed time; the close keeps the frame's own receipt time. Codex reasoning rows likewise carry their item/started time. * perf(native-chat): stop rewriting the reasoning row for every thinking token Claude sends a thinking_tokens frame after every thinking delta. Every frame the stream path did not consume forced the streamed text out first, bypassing the checkpoint widening and the coalescing window, so a single thinking block was rewritten and republished once per token: 212 full-row writes and 258 KB for one captured block. A frame that can write no row (token tallies, stream deltas no registry carries, pings) no longer forces that flush; every frame that can write still does. The same block now takes 14 writes. * fix(native-chat): time Codex reasoning by receipt and close evicted rows Codex summary text lands 13-50 ms before item/completed, inside one coalescing window, so a reasoning row's first write can be its completion. Item boundaries are now stamped with their host receipt time the way turn boundaries already were, so a retried or buffered delivery keeps it, and the item translator falls back to the host clock rather than Date.now(). The row carries its item/started time on every first write, including the completion, so its span is item/started to item/completed. An evicted active item is now closed from the text streamed so far, like both settle paths, and the eviction runs before the incoming item is tracked: tracking first let the stream bound drop the evictee's text before the eviction could close its row, stranding it running. completedAt now has one meaning everywhere: the host time the message was seen to end, or the end of the turn or stream that cut it off; absent only when no end was seen live. * refactor(native-chat): keep the Codex streaming body translation pure The streaming translation preserves a reasoning body as-is again; the stream writer, which is what knows the item has not completed, stamps it running. * test(native-chat): keep a re-collapsed reasoning row collapsed through a revision * fix(native-chat): estimate a collapsed reasoning row as its trigger A reasoning row renders collapsed, as one small button, but its height was estimated from its full text: a 4,129-character summary reserved about 950 px for a 24 px row, so long chats jumped as rows were measured. It is now estimated at the trigger's height; opening the row remeasures it. * fix(native-chat): probe the CLI the launch will run, and learn from a refusal The thinking-display gate probed `claude --version` with Orca's own cwd and env, while the launch spawns with the workspace's cwd and the shell env. Behind a version manager's shim those can pick different CLIs, so the probe could approve a CLI the launch never ran, and an older CLI exits on the unknown flag before the session starts. The probe also only counted if it had already finished when resolution did, so a first launch, or the first after a CLI update, usually went without the flag. The probe now runs with the launch's own resolved cwd and env, is keyed by the binary and the workspace, and a launch waits up to 200 ms for it (about 3x the probe's measured p95) before going without the flag. Only answers are kept, so a slow or failed probe is asked again next launch. A child that exits with commander's "unknown option '--thinking-display'" marks that binary in that workspace so the next launch skips the flag; that one start fails exactly as any CLI startup failure does today. * test(native-chat): type the thinking-display probe mock with both of its parameters * fix(native-chat): recheck the account switch after the probe, last as before Moving the invocation ahead of the probe, the transcript check and the permission mode put its account-switch recheck before those awaits, so a switch that began during them launched unchecked. The invocation is the last await again. The probe gets its own env from the same sources the launch uses, the inherited env and the overlay with the CLI's runtime on PATH, built by the same code, minus every credential: asking a CLI its version needs none. * fix(native-chat): wait up to 1.5 s for a cold probe, and remember every outcome 200 ms only covered a warm CLI; a cold disk, a node install or an antivirus scan exceeds it, and that chat's child then ran its whole life without summaries. A launch now waits up to 1.5 s, once per binary per workspace, measured from when that binary's probe began, so a later launch never waits again on a probe already past it. The probe gets its own 10 s kill timeout, and every outcome, including no version printed, a failure or a kill, is kept for the binary's life, so a probe that hangs costs one launch rather than every one. A refusal seen while a probe still runs wins over its late answer. * fix(native-chat): leave Thinking to the activity line while reasoning runs A reasoning row still being written drew a pulsing "Thinking…" header right under the turn's activity line, which already says Thinking: two live indicators for one fact. In a running turn an open reasoning row now draws nothing and reserves no height; it appears when it closes, as "Thought for N s". A row from a host that keeps no state, a closed row, and a row left open by a turn that ended draw as before. The row's Thinking headline is gone with its catalog key. * fix(native-chat): end every unfinished Codex item through one rule A reasoning completion with no text of its own left the row its stream wrote running for good: the completion translated to nothing and the item left the active set, so no settle could reach it. It now closes from the text streamed so far. Settlement, eviction and that completion now build an unfinished item's row through one choice (the streamed text when there is any, else the item as it started) and end it through one rule. An evicted file change with streamed tool output no longer keeps that output as its patch; it reads as interrupted, as a settled one does. A completion whose start was never recorded claims no span, so it reads "Thought". * fix(native-chat): catalogue Claude's stream keep-alive as benign An uncatalogued `ping` stream frame classified as substantive, so the fallback wrote a visible "claude · message:stream_event:ping" row, and since such frames no longer force streamed text out first, a ping inside a coalescing window landed above the open reasoning row. A ping is now benign: it writes no row. * fix(native-chat): bound finding the binary by the probe budget, and keep it LRU Resolving the command's real path and its mtime was awaited before the budgeted wait, so a slow filesystem could hold a launch indefinitely; it now counts against the same budget, and running out caches nothing. The cache is least recently used rather than first written, and holds 32 binary-and-workspace entries rather than 16. * feat(mobile): collapse reasoning rows the way desktop does With summaries on, every Claude turn now carries reasoning text, and the phone drew all of it inline, dimmed, between the prompt and the answer. Mobile now draws a reasoning row as desktop does: collapsed to "Thought for N s", "Thought" or "Reasoning", its text mounted only once opened, and nothing at all while the row is still being written in the live turn or has no text. The headline and the visibility rule live in one shared module both clients read, so they cannot drift. * fix(native-chat): let a failed CLI probe heal instead of latching A probe killed at its timeout, failing to spawn under a loaded boot, or printing no version was cached as "no flag" for the binary's life, so that workspace never got summaries again in that run. Only a version (either side of the floor) or the CLI's own refusal is kept for good now; a probe that gave no version is kept for 10 minutes, so a hung CLI still costs one wait per stretch and a boot-time failure heals. * docs(native-chat): say exactly what the version probe's env leaves out * fix(native-chat): record a Codex item's start whatever its first frame carried The start was recorded only for an item/started that wrote no row, so a reasoning item that started with text lost it and its completion claimed no span. Every tracked item/started now records its receipt time, and the started write carries it too. * fix(mobile): label the reasoning toggle and give it a full touch target The toggle now tells a screen reader what it is, "Reasoning: Thought for 3s", as desktop's prefix does, and reaches a 44 pt target. The shared English copy stays private to the module that formats it. * refactor(native-chat): build reasoning rows from one provider-neutral helper Claude, Codex and the terminal sweeps each built the reasoning row body and its running/ended stamp themselves. They now share journal-reasoning-row: blank text journals no row, text is bounded the same way, and an end carries completedAt only when the host saw it. * refactor(codex): move the active journal item type into the contracts file codex-unfinished-item-body imported the type from the settlement module, which imports values from it. * test(claude): read the launch PATH the way Windows spells it * test(claude): compare the probe's PATH to the launch's without Orca's CLI dir When the CLI's directory also holds node (Linux CI's /usr/local/bin), the runtime pairing puts that directory first, ahead of the Orca CLI directory the launch adds, so the launch PATH no longer ends with the probe's. Both still resolve the same claude and shims. The test now checks that exactly, for a CLI with and without a sibling node. * feat(native-chat): lead the reasoning row with a brain glyph in the tool-row column * fix(native-chat): route the reasoning glyph through the shared icon names, keep its chevron findable, and match it on mobile * fix(mobile): keep the long-press actions sheet on reasoning rows for Android * fix(native-chat): forward every exit argument through the thinking-display connection wrapper * refactor(codex): keep the receipt-timed notification methods with the event they stamp * feat(native-chat): read an open reasoning block through the one live "Thinking" line While the agent's open reasoning block has text, the turn's live activity line is its disclosure: collapsed by default, expandable to the live text (capped and scrollable), and the block's row draws nothing meanwhile. When the block ends, its row appears in place, open if the reader opened it live, because the line and the row read one disclosure key. Which block the line discloses is derived from the line's own render condition, so a row is never hidden while nothing on screen shows it; any other open block (a subagent's, or one a prompt pushed off the line) draws as "Reasoning". Desktop and mobile alike; no host or wire change. * fix(native-chat): a slot kept for its turn bar or diff rollup no longer draws its message The transcript row drew the message of every message slot, while the slot builder pushes a slot for a row it declined to draw whenever that row also carries its turn's bar or diff rollup. So the open reasoning block the live line discloses still drew as a "Reasoning" row when it was a provider-opened turn's first row or the last row of a turn that changed files, and one click opened both. The builder's decision now travels on the slot (`drawsMessage`) and the row draws only the bar and rollup when it is false; the row-level `folded` guard it made redundant is gone. * refactor(native-chat): draw the live line from one shared value, with one live region Desktop and mobile now render the live activity line from one pure function, `nativeChatLiveLine`: whether it draws, what it says, and the open reasoning block it discloses with the text it has so far. The open block's row is hidden from that same value, so desktop no longer restates the line's render condition beside it, and the lines no longer re-derive the text. The line keeps one element, and so one live region, through every state; only its trigger and body come and go, so a screen reader hears "Thinking" and the label after it. On mobile the live text gets the finished row's Android long press (copy or select through the message actions sheet), the line's touch target is the row's 44 pt, and its label and body sit in the finished row's column so nothing moves when the row takes over. * fix(mobile): keep the live text's actions sheet on the block it was opened for On Android the sheet opened by a long press on the live reasoning was a flag gated on a live block: it vanished when the block ended, mid Select text, and the stale flag reopened it unprompted on the next block. The sheet now holds the message it was opened for. * test(native-chat): the reasoning body owns its tone, live and once landed Rendered QA on a pre-merge build showed the live line's open reasoning in full foreground and the landed row's in muted text, so it dimmed as the block landed: the body set no colour of its own and inherited one from wherever it was mounted. Since the main merge (017ad743fa5) the body carries the chat's faint tone itself; this pins that, and that the line and the row draw the same body. --------- Co-authored-by: Merge Sim --- .../MobileNativeChatLiveLine.android.test.ts | 123 +++++++ .../session/MobileNativeChatLiveLine.test.ts | 153 +++++++++ .../src/session/MobileNativeChatLiveLine.tsx | 116 +++++++ .../MobileNativeChatLongPressContent.tsx | 22 ++ .../MobileNativeChatMessage.android.test.ts | 33 ++ .../session/MobileNativeChatMessage.test.ts | 86 +++++ .../src/session/MobileNativeChatMessage.tsx | 73 +++-- .../session/MobileNativeChatReasoningRow.tsx | 102 ++++++ .../MobileNativeChatTurnStatus.test.ts | 40 +-- .../session/MobileNativeChatTurnStatus.tsx | 27 +- .../src/session/MobileNativeChatView.test.ts | 1 + mobile/src/session/MobileNativeChatView.tsx | 20 +- .../MobileNativeChatView.turn-status.test.ts | 9 +- .../MobileNativeChatView.waiting-rows.test.ts | 2 +- .../mobile-native-chat-message-styles.ts | 26 +- ...obile-native-chat-turn-disclosure.test.tsx | 68 ++++ .../use-mobile-native-chat-turn-disclosure.ts | 122 +++++-- src/main/claude/claude-hook-event-versions.ts | 12 +- src/main/claude/claude-message-journaling.ts | 31 +- src/main/claude/claude-open-turn.ts | 6 + .../claude/claude-streamed-block-identity.ts | 48 +-- .../claude-streamed-text-checkpoints.ts | 28 +- src/main/claude/claude-streamed-thinking.ts | 156 +++++++++ ...-journal-translation-context-usage.test.ts | 7 +- ...ude-structured-journal-translation.test.ts | 3 +- .../claude-structured-journal-translation.ts | 45 ++- ...laude-structured-launch-resolution.test.ts | 89 ++++- .../claude-structured-launch-resolution.ts | 144 +++++--- .../claude-structured-provider-fallback.ts | 19 ++ ...ude-structured-reasoning-lifecycle.test.ts | 213 ++++++++++++ .../claude-structured-reasoning.test.ts | 218 +++++++++++++ .../claude-thinking-display-support.test.ts | 168 ++++++++++ .../claude/claude-thinking-display-support.ts | 175 ++++++++++ ...codex-persistent-command-retention.test.ts | 2 + .../codex-structured-item-stream-bounds.ts | 16 + .../codex-structured-item-stream-contracts.ts | 5 +- .../codex/codex-structured-item-streams.ts | 29 +- .../codex-structured-item-translation.test.ts | 28 ++ .../codex-structured-item-translation.ts | 22 +- .../codex-structured-journal-contracts.ts | 13 + .../codex/codex-structured-journal-items.ts | 118 ++++--- .../codex-structured-journal-settlement.ts | 59 +--- .../codex/codex-structured-journal-sink.ts | 3 +- ...red-journal-translation-settlement.test.ts | 3 +- ...red-journal-translation-turn-boundaries.ts | 1 + .../codex-structured-journal-translation.ts | 3 +- ...dex-structured-reasoning-lifecycle.test.ts | 267 +++++++++++++++ .../codex/codex-structured-session-acquire.ts | 7 +- ...ructured-session-adapter-lifecycle.test.ts | 23 ++ .../codex/codex-structured-session-state.ts | 12 +- src/main/codex/codex-unfinished-item-body.ts | 72 ++++ .../journal-reasoning-row.ts | 39 +++ .../journal-terminal-settlement.ts | 11 + .../claude-stream-json-frame-schema.ts | 2 + .../provider-frame-disposition.test.ts | 3 + .../provider-frame-disposition.ts | 1 + ...gent-session-dead-generation-settlement.ts | 8 +- ...ured-agent-session-reasoning-sweep.test.ts | 110 +++++++ ...ude-structured-session-integration.test.ts | 13 + .../runtime/orca-runtime-get-worktree-ps.ts | 3 + .../structured-agent-runtime-registrations.ts | 1 + .../structured-agent-session-runtime.ts | 3 + .../structured-claude-runtime-adapter.ts | 39 ++- ...ed-claude-thinking-display-refusal.test.ts | 97 ++++++ ...iveChatMessageList.live-reasoning.test.tsx | 308 ++++++++++++++++++ .../native-chat/NativeChatMessageList.tsx | 59 ++-- .../native-chat/NativeChatMessageRow.test.tsx | 15 +- .../native-chat/NativeChatMessageRow.tsx | 29 +- .../NativeChatReasoningDisclosure.tsx | 45 +++ .../NativeChatReasoningRow.test.tsx | 217 ++++++++++++ .../native-chat/NativeChatReasoningRow.tsx | 77 +++++ .../native-chat/NativeChatToolIcon.tsx | 4 +- .../native-chat/NativeChatTranscriptRow.tsx | 3 +- .../NativeChatTurnActivityLine.tsx | 80 ++++- .../native-chat-row-height-estimate.test.ts | 12 + .../native-chat-row-height-estimate.ts | 9 +- .../native-chat-subagent-section-slots.ts | 1 + .../native-chat-subagent-sections.test.ts | 11 + .../native-chat-transcript-slots.test.ts | 15 + .../native-chat-transcript-slots.ts | 28 +- .../use-native-chat-transcript-slots.ts | 87 +++++ ...ve-chat-transcript-window.options.test.tsx | 1 + src/renderer/src/i18n/locales/en.json | 3 + src/renderer/src/i18n/locales/es.json | 3 + src/renderer/src/i18n/locales/fr.json | 3 + src/renderer/src/i18n/locales/ja.json | 3 + src/renderer/src/i18n/locales/ko.json | 3 + src/renderer/src/i18n/locales/zh.json | 3 + .../agent-session-journal-schemas.test.ts | 22 ++ src/shared/agent-session-journal-schemas.ts | 26 +- ...gent-session-journal-thread-goal-schema.ts | 22 ++ src/shared/agent-session-journal-types.ts | 12 + src/shared/native-chat-live-line.test.ts | 55 ++++ src/shared/native-chat-live-line.ts | 38 +++ src/shared/native-chat-reasoning-row.test.ts | 98 ++++++ src/shared/native-chat-reasoning-row.ts | 98 ++++++ src/shared/native-chat-tool-icon.ts | 2 + src/shared/native-chat-turn-membership.ts | 10 + src/shared/native-chat-types.ts | 5 + ...structured-agent-session-live-turn.test.ts | 14 + .../structured-agent-session-live-turn.ts | 4 +- .../structured-agent-session-projection.ts | 18 + ...agent-session-reasoning-projection.test.ts | 55 ++++ 103 files changed, 4418 insertions(+), 478 deletions(-) create mode 100644 mobile/src/session/MobileNativeChatLiveLine.android.test.ts create mode 100644 mobile/src/session/MobileNativeChatLiveLine.test.ts create mode 100644 mobile/src/session/MobileNativeChatLiveLine.tsx create mode 100644 mobile/src/session/MobileNativeChatLongPressContent.tsx create mode 100644 mobile/src/session/MobileNativeChatReasoningRow.tsx create mode 100644 src/main/claude/claude-streamed-thinking.ts create mode 100644 src/main/claude/claude-structured-reasoning-lifecycle.test.ts create mode 100644 src/main/claude/claude-structured-reasoning.test.ts create mode 100644 src/main/claude/claude-thinking-display-support.test.ts create mode 100644 src/main/claude/claude-thinking-display-support.ts create mode 100644 src/main/codex/codex-structured-reasoning-lifecycle.test.ts create mode 100644 src/main/codex/codex-unfinished-item-body.ts create mode 100644 src/main/native-chat/agent-session-journal/journal-reasoning-row.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-reasoning-sweep.test.ts create mode 100644 src/main/runtime/structured-claude-thinking-display-refusal.test.ts create mode 100644 src/renderer/src/components/native-chat/NativeChatMessageList.live-reasoning.test.tsx create mode 100644 src/renderer/src/components/native-chat/NativeChatReasoningDisclosure.tsx create mode 100644 src/renderer/src/components/native-chat/NativeChatReasoningRow.test.tsx create mode 100644 src/renderer/src/components/native-chat/NativeChatReasoningRow.tsx create mode 100644 src/renderer/src/components/native-chat/use-native-chat-transcript-slots.ts create mode 100644 src/shared/agent-session-journal-thread-goal-schema.ts create mode 100644 src/shared/native-chat-live-line.test.ts create mode 100644 src/shared/native-chat-live-line.ts create mode 100644 src/shared/native-chat-reasoning-row.test.ts create mode 100644 src/shared/native-chat-reasoning-row.ts create mode 100644 src/shared/structured-agent-session-reasoning-projection.test.ts diff --git a/mobile/src/session/MobileNativeChatLiveLine.android.test.ts b/mobile/src/session/MobileNativeChatLiveLine.android.test.ts new file mode 100644 index 00000000000..daecbe6ab62 --- /dev/null +++ b/mobile/src/session/MobileNativeChatLiveLine.android.test.ts @@ -0,0 +1,123 @@ +import { createElement, type ReactNode } from 'react' +import { act, create, type ReactTestInstance, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { NativeChatLiveReasoning } from '../../../src/shared/native-chat-reasoning-row' + +// Why a separate file: Android has no inline selection, so the live text is copied the way a +// finished row's is, through the message actions sheet. +vi.mock('react-native', async () => { + const React = await import('react') + const host = + (name: string) => + ({ children, ...props }: { children?: ReactNode }): ReactNode => + React.createElement(name, props, children) + return { + ActivityIndicator: host('ActivityIndicator'), + Platform: { OS: 'android' }, + Pressable: host('Pressable'), + Text: host('Text'), + View: host('View'), + StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } + } +}) +vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' })) +vi.mock('./MobileNativeChatReasoningRow', () => ({ + MobileNativeChatReasoningBody: 'ReasoningBody' +})) +vi.mock('./MobileNativeChatMessageActionsSheet', () => ({ + MobileNativeChatMessageActionsSheet: 'MessageActionsSheet' +})) + +import { MobileNativeChatLiveLine } from './MobileNativeChatLiveLine' + +const block: NativeChatLiveReasoning = { + message: { + id: 'r-1', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }], + timestamp: null, + source: 'transcript', + state: 'running' + }, + markdown: 'Weighing two approaches' +} + +describe('MobileNativeChatLiveLine on Android', () => { + let renderer: ReactTestRenderer | null = null + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + }) + + const byType = (type: string): ReactTestInstance[] => + renderer!.root.findAll((node) => String(node.type) === type) + + it('opens the actions sheet for the live block from a long press on its text', () => { + act(() => { + renderer = create( + createElement(MobileNativeChatLiveLine, { + line: { thinking: true, activityText: null, reasoning: block, reasoningExpanded: true }, + onToggleReasoning: vi.fn(), + fontScale: 1 + }) + ) + }) + expect(byType('MessageActionsSheet')).toHaveLength(0) + const [body] = byType('ReasoningBody') + expect(typeof body?.props.onLongPress).toBe('function') + + act(() => body!.props.onLongPress()) + const [sheet] = byType('MessageActionsSheet') + // The sheet copies and selects the message's own text: the block as it stands. + expect(sheet?.props.message).toBe(block.message) + + act(() => sheet!.props.onClose()) + expect(byType('MessageActionsSheet')).toHaveLength(0) + }) + + it('keeps the sheet through the block ending, and never opens one for the next by itself', () => { + const line = (reasoning: NativeChatLiveReasoning | null) => + createElement(MobileNativeChatLiveLine, { + line: { thinking: true, activityText: null, reasoning, reasoningExpanded: true }, + onToggleReasoning: vi.fn(), + fontScale: 1 + }) + act(() => { + renderer = create(line(block)) + }) + act(() => byType('ReasoningBody')[0]!.props.onLongPress()) + // The block ends while the reader copies or selects it. + act(() => renderer!.update(line(null))) + expect(byType('MessageActionsSheet').map((sheet) => sheet.props.message)).toEqual([ + block.message + ]) + act(() => byType('MessageActionsSheet')[0]!.props.onClose()) + const next: NativeChatLiveReasoning = { + message: { ...block.message, id: 'r-2' }, + markdown: 'Next thought' + } + act(() => renderer!.update(line(next))) + expect(byType('MessageActionsSheet')).toHaveLength(0) + }) + + it('opens no sheet for the next block when the first ends with the sheet open', () => { + const line = (reasoning: NativeChatLiveReasoning | null) => + createElement(MobileNativeChatLiveLine, { + line: { thinking: true, activityText: null, reasoning, reasoningExpanded: true }, + onToggleReasoning: vi.fn(), + fontScale: 1 + }) + act(() => { + renderer = create(line(block)) + }) + act(() => byType('ReasoningBody')[0]!.props.onLongPress()) + act(() => renderer!.update(line(null))) + const next: NativeChatLiveReasoning = { + message: { ...block.message, id: 'r-2' }, + markdown: 'Next thought' + } + act(() => renderer!.update(line(next))) + // Still the first block's sheet, the one the reader opened; none for r-2. + expect(byType('MessageActionsSheet').map((sheet) => sheet.props.message.id)).toEqual(['r-1']) + }) +}) diff --git a/mobile/src/session/MobileNativeChatLiveLine.test.ts b/mobile/src/session/MobileNativeChatLiveLine.test.ts new file mode 100644 index 00000000000..37bcc460ca5 --- /dev/null +++ b/mobile/src/session/MobileNativeChatLiveLine.test.ts @@ -0,0 +1,153 @@ +import { createElement, type ReactNode } from 'react' +import { act, create, type ReactTestInstance, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { NativeChatLiveReasoning } from '../../../src/shared/native-chat-reasoning-row' + +const pressableMounts = vi.hoisted(() => ({ count: 0 })) + +vi.mock('react-native', async () => { + const React = await import('react') + const host = + (name: string) => + ({ children, ...props }: { children?: ReactNode }): ReactNode => + React.createElement(name, props, children) + // Counts mounts, so a test can tell the live region was kept rather than replaced. + const Pressable = ({ children, ...props }: { children?: ReactNode }): ReactNode => { + React.useEffect(() => { + pressableMounts.count += 1 + }, []) + return React.createElement('Pressable', props, children) + } + return { + ActivityIndicator: host('ActivityIndicator'), + Platform: { OS: 'ios' }, + Pressable, + Text: host('Text'), + View: host('View'), + StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } + } +}) +vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' })) +vi.mock('./MobileNativeChatReasoningRow', () => ({ + MobileNativeChatReasoningBody: 'ReasoningBody' +})) +vi.mock('./MobileNativeChatMessageActionsSheet', () => ({ + MobileNativeChatMessageActionsSheet: 'MessageActionsSheet' +})) + +import { MobileNativeChatLiveLine } from './MobileNativeChatLiveLine' + +const block: NativeChatLiveReasoning = { + message: { + id: 'r-1', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }], + timestamp: null, + source: 'transcript', + state: 'running' + }, + markdown: 'Weighing two approaches' +} + +describe('MobileNativeChatLiveLine', () => { + let renderer: ReactTestRenderer | null = null + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + pressableMounts.count = 0 + }) + + const onToggleReasoning = vi.fn() + function element( + fields: { + thinking?: boolean + activityText?: string | null + reasoning?: NativeChatLiveReasoning | null + reasoningExpanded?: boolean + } = {} + ) { + return createElement(MobileNativeChatLiveLine, { + line: { + thinking: true, + activityText: null, + reasoning: null, + reasoningExpanded: false, + ...fields + }, + onToggleReasoning, + fontScale: 1 + }) + } + function render(fields: Parameters[0] = {}): ReactTestInstance { + act(() => { + renderer = create(element(fields)) + }) + return renderer!.root + } + const byType = (root: ReactTestInstance, type: string): ReactTestInstance[] => + root.findAll((node) => String(node.type) === type) + const labels = (root: ReactTestInstance): string[] => + byType(root, 'Text').map((text) => String(text.children.join(''))) + const header = (root: ReactTestInstance): ReactTestInstance => + root.find((node) => String(node.type) === 'Pressable') + + it('reads "Thinking" beside one spinner while the turn reasons', () => { + const root = render() + expect(labels(root)).toEqual(['Thinking']) + expect(byType(root, 'ActivityIndicator')).toHaveLength(1) + }) + + // The bar owns the clock; the tail line never repeats it. + it('reads plain "Working…" when the turn is not reasoning, and holds no timer', () => { + vi.useFakeTimers() + const root = render({ thinking: false }) + expect(labels(root)).toEqual(['Working…']) + expect(vi.getTimerCount()).toBe(0) + vi.useRealTimers() + }) + + it('lets provider activity text beat both fallbacks', () => { + expect(labels(render({ activityText: 'Running pnpm test' }))).toEqual(['Running pnpm test']) + }) + + it('announces what it says to assistive tech, and is no button while it discloses nothing', () => { + const line = header(render()) + expect(line.props.accessibilityLiveRegion).toBe('polite') + expect(line.props.accessibilityLabel).toBe('Thinking') + expect(line.props.accessibilityRole).toBeUndefined() + expect(line.props.onPress).toBeUndefined() + }) + + it('discloses the open block under one "Thinking", collapsed, toggled by its block key', () => { + const root = render({ reasoning: block }) + expect(labels(root)).toEqual(['Thinking']) + const line = header(root) + expect(line.props.accessibilityRole).toBe('button') + expect(line.props.accessibilityState).toEqual({ expanded: false }) + expect(byType(root, 'ReasoningBody')).toHaveLength(0) + act(() => line.props.onPress()) + expect(onToggleReasoning).toHaveBeenCalledWith('reasoning:r-1') + }) + + it('shows the live text outside the live region once opened', () => { + const root = render({ reasoning: block, reasoningExpanded: true }) + const [body] = byType(root, 'ReasoningBody') + expect(body?.props.markdown).toBe('Weighing two approaches') + let ancestor = body?.parent ?? null + while (ancestor) { + expect(ancestor.props.accessibilityLiveRegion).toBeUndefined() + ancestor = ancestor.parent + } + // iOS selects inline, so the body takes no long press. + expect(body?.props.onLongPress).toBeUndefined() + }) + + it('keeps one live region while it turns into the disclosure and back', () => { + render({ thinking: false }) + act(() => renderer!.update(element({ reasoning: block }))) + expect(header(renderer!.root).props.accessibilityLabel).toBe('Thinking') + act(() => renderer!.update(element({ thinking: false }))) + expect(header(renderer!.root).props.accessibilityLabel).toBe('Working…') + expect(pressableMounts.count).toBe(1) + }) +}) diff --git a/mobile/src/session/MobileNativeChatLiveLine.tsx b/mobile/src/session/MobileNativeChatLiveLine.tsx new file mode 100644 index 00000000000..30ead56325c --- /dev/null +++ b/mobile/src/session/MobileNativeChatLiveLine.tsx @@ -0,0 +1,116 @@ +import { useCallback, useState } from 'react' +import { ActivityIndicator, Pressable, StyleSheet, Text, View } from 'react-native' +import { ChevronRight } from 'lucide-react-native' +import { nativeChatReasoningDisclosureKey } from '../../../src/shared/native-chat-reasoning-row' +import { formatNativeChatActiveTurnLabel } from '../../../src/shared/native-chat-turn-status' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { INLINE_TEXT_SELECTION } from '../components/inline-text-selection' +import { colors, spacing, typography } from '../theme/mobile-theme' +import { MobileNativeChatMessageActionsSheet } from './MobileNativeChatMessageActionsSheet' +import { MobileNativeChatReasoningBody } from './MobileNativeChatReasoningRow' +import type { MobileNativeChatLiveLine as LiveLine } from './use-mobile-native-chat-turn-disclosure' + +/** The live turn's tail line: a spinner beside what the provider says it is doing, else + * "Thinking", else "Working…". The clock stays in the turn bar. While the agent's open reasoning + * block has text it is also that block's disclosure, and the block's row draws nothing. + * Desktop parity: `NativeChatTurnActivityLine`. */ +export function MobileNativeChatLiveLine({ + line, + onToggleReasoning, + fontScale, + onOpenFile +}: { + line: LiveLine + onToggleReasoning: (key: string) => void + fontScale: number + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const { reasoning, reasoningExpanded } = line + const open = reasoning !== null && reasoningExpanded + const label = formatNativeChatActiveTurnLabel(line) + // Android has no inline selection; the finished row's long-press sheet copies the live text too. + // It holds the block it opened for, so it outlives that block ending and never reopens by itself. + const [actionsFor, setActionsFor] = useState(null) + const liveMessage = reasoning?.message ?? null + const openActions = useCallback(() => setActionsFor(liveMessage), [liveMessage]) + return ( + + {/* One element for every state of the line, so TalkBack hears each new label; the body + sits outside it. */} + [styles.row, reasoning && pressed && styles.pressed]} + onPress={ + reasoning + ? () => onToggleReasoning(nativeChatReasoningDisclosureKey(reasoning.message.id)) + : undefined + } + // With the row's 32 pt height, the 44 pt target the reasoning row has. + hitSlop={6} + accessibilityRole={reasoning ? 'button' : undefined} + accessibilityState={reasoning ? { expanded: open } : undefined} + accessibilityLabel={label} + accessibilityLiveRegion="polite" + > + + + + + {label} + + {reasoning ? ( + + + + ) : null} + + {reasoning && open ? ( + + + + ) : null} + {actionsFor ? ( + setActionsFor(null)} + /> + ) : null} + + ) +} + +// The finished reasoning row's geometry (row gutter, 15 pt glyph slot, its gap, its 32 pt height), +// so the label and the open text do not move when that row takes over. +const styles = StyleSheet.create({ + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm, + minHeight: 32, + paddingHorizontal: spacing.lg + }, + // The spinner is wider than the brain; centred in the brain's slot it keeps the text column. + glyph: { + width: 15, + alignItems: 'center', + overflow: 'visible' + }, + pressed: { + opacity: 0.6 + }, + label: { + color: colors.textMuted, + fontSize: typography.bodySize, + flexShrink: 1 + }, + caretOpen: { + transform: [{ rotate: '90deg' }] + }, + body: { + paddingHorizontal: spacing.lg + } +}) diff --git a/mobile/src/session/MobileNativeChatLongPressContent.tsx b/mobile/src/session/MobileNativeChatLongPressContent.tsx new file mode 100644 index 00000000000..923e949b7b4 --- /dev/null +++ b/mobile/src/session/MobileNativeChatLongPressContent.tsx @@ -0,0 +1,22 @@ +import type { ComponentProps, ReactNode } from 'react' +import { Pressable, View } from 'react-native' + +/** A message body that opens the actions sheet on long press (Android, which has no inline + * selection). Keep the existing responder hierarchy on platforms with inline selection. */ +export function MobileNativeChatLongPressContent({ + onLongPress, + style, + children +}: { + onLongPress?: () => void + style: ComponentProps['style'] + children: ReactNode +}): React.JSX.Element { + return onLongPress ? ( + + {children} + + ) : ( + {children} + ) +} diff --git a/mobile/src/session/MobileNativeChatMessage.android.test.ts b/mobile/src/session/MobileNativeChatMessage.android.test.ts index fff2a8d9eba..104aa35d2a0 100644 --- a/mobile/src/session/MobileNativeChatMessage.android.test.ts +++ b/mobile/src/session/MobileNativeChatMessage.android.test.ts @@ -23,6 +23,8 @@ vi.mock('react-native', async () => { Image: 'Image', Platform: { OS: 'android' }, Pressable: 'Pressable', + ScrollView: ({ children, ...props }: { children?: ReactNode }) => + React.createElement('ScrollView', props, children), Text, View: ({ children, ...props }: { children?: ReactNode }) => React.createElement('View', props, children), @@ -32,6 +34,7 @@ vi.mock('react-native', async () => { vi.mock('expo-clipboard', () => ({ setStringAsync: vi.fn() })) vi.mock('lucide-react-native', () => ({ ArrowUp: 'ArrowUp', + Brain: 'Brain', ChevronDown: 'ChevronDown', Copy: 'Copy', SquareChevronRight: 'SquareChevronRight', @@ -106,6 +109,36 @@ describe('MobileNativeChatMessage on Android', () => { expect(byType('MessageActionsSheet')).toHaveLength(1) }) + it('opens the actions sheet from a long press on an expanded reasoning row, not on its headline', () => { + const reasoning: NativeChatMessage = { + ...message, + id: 'r1', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }], + timestamp: 1_000, + state: 'completed', + completedAt: 4_000 + } + act(() => { + renderer = create(createElement(MobileNativeChatMessage, { message: reasoning })) + }) + const [toggle] = byType('Pressable') + // A long press on the headline stays a plain toggle tap. + expect(toggle!.props.onLongPress).toBeUndefined() + act(() => toggle!.props.onPress()) + + const body = byType('Pressable').find((node) => node.props.onLongPress !== undefined) + expect(typeof body?.props.onLongPress).toBe('function') + const [markdown] = byType('MobileMarkdown') + expect(markdown!.props.onLongPress).toBe(body!.props.onLongPress) + + act(() => body!.props.onLongPress()) + const [sheet] = byType('MessageActionsSheet') + expect(sheet!.props.message).toBe(reasoning) + act(() => sheet!.props.onClose()) + expect(byType('MessageActionsSheet')).toHaveLength(0) + }) + it('renders the user bubble without inline selection', () => { act(() => { renderer = create( diff --git a/mobile/src/session/MobileNativeChatMessage.test.ts b/mobile/src/session/MobileNativeChatMessage.test.ts index 91248061048..2a2489bbdbc 100644 --- a/mobile/src/session/MobileNativeChatMessage.test.ts +++ b/mobile/src/session/MobileNativeChatMessage.test.ts @@ -26,6 +26,8 @@ vi.mock('react-native', async () => { Image: 'Image', Platform: { OS: 'ios' }, Pressable: 'Pressable', + ScrollView: ({ children, ...props }: { children?: unknown }) => + React.createElement('ScrollView', props, children), Text, View: ({ children, ...props }: { children?: unknown }) => React.createElement('View', props, children), @@ -35,6 +37,7 @@ vi.mock('react-native', async () => { vi.mock('expo-clipboard', () => ({ setStringAsync: vi.fn() })) vi.mock('lucide-react-native', () => ({ ArrowUp: 'ArrowUp', + Brain: 'Brain', ChevronDown: 'ChevronDown', Copy: 'Copy', SquareChevronRight: 'SquareChevronRight', @@ -79,6 +82,7 @@ describe('MobileNativeChatMessage', () => { workedSeconds: number | null } | null onToggleTurn?: () => void + reasoningIsLive?: boolean } = {} ): ReactTestRenderer { act(() => { @@ -375,4 +379,86 @@ describe('MobileNativeChatMessage', () => { expect(textIn(tree.root)).toEqual(['go']) }) }) + + describe('a reasoning row', () => { + const reasoning = (fields: Partial = {}): NativeChatMessage => ({ + id: 'r1', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }], + timestamp: 1_000, + source: 'transcript', + state: 'completed', + completedAt: 4_000, + ...fields + }) + const toggleOf = (tree: ReactTestRenderer): ReactTestInstance => + tree.root.find( + (node) => String(node.type) === 'Pressable' && node.props.accessibilityRole === 'button' + ) + const markdownIn = (tree: ReactTestRenderer): ReactTestInstance[] => + tree.root.findAll((node) => String(node.type) === 'MobileMarkdown') + + it('starts collapsed to its headline, with its text unmounted', () => { + const tree = render(reasoning()) + expect(textIn(tree.root)).toContain('Thought for 3s') + expect(toggleOf(tree).props.accessibilityState).toEqual({ expanded: false }) + // Said with what it is, as desktop's screen-reader prefix does, on a 32 + 2 × 6 pt target. + expect(toggleOf(tree).props.accessibilityLabel).toBe('Reasoning: Thought for 3s') + expect(toggleOf(tree).props.hitSlop).toBe(6) + expect(markdownIn(tree)).toHaveLength(0) + }) + + it('leads its headline with the brain, as desktop does', () => { + const [first] = toggleOf(render(reasoning())).children + expect(typeof first === 'string' ? first : first?.type).toBe('Brain') + }) + + it('mounts its text once opened', () => { + const tree = render(reasoning()) + act(() => toggleOf(tree).props.onPress()) + expect(toggleOf(tree).props.accessibilityState).toEqual({ expanded: true }) + expect(markdownIn(tree).map((node) => node.props.content)).toEqual([ + 'Weighing two approaches' + ]) + }) + + it('draws nothing while the live line discloses it, or when blank', () => { + expect( + render(reasoning({ state: 'running' }), { + activeTurnIsWorking: true, + reasoningIsLive: true + }).toJSON() + ).toBeNull() + expect(render(reasoning({ blocks: [{ type: 'text', text: ' \n ' }] })).toJSON()).toBeNull() + }) + + // The turn's bar is not the block's: hiding the block must not hide the bar it sits on. + it("still draws its turn's bar while the live line discloses it", () => { + const tree = render(reasoning({ state: 'running' }), { + activeTurnIsWorking: true, + reasoningIsLive: true, + turnStatus: { startedAt: 1_000, thinking: true, workedSeconds: null } + }) + expect(tree.root.findAll((node) => String(node.type) === 'Pressable')).toHaveLength(0) + expect(textIn(tree.root).some((text) => text.startsWith('Working for'))).toBe(true) + }) + + // Only the block the line discloses hides: a subagent's or a stale open block draws, unended. + it('draws any other open block in its working turn as Reasoning', () => { + const child = render(reasoning({ state: 'running', agentId: 'sub-1' }), { + activeTurnIsWorking: true + }) + expect(textIn(child.root)).toContain('Reasoning') + expect(toggleOf(child).props.accessibilityLabel).toBe('Reasoning') + }) + + it('says only what the host saw', () => { + expect(textIn(render(reasoning({ state: 'running' })).root)).toContain('Thought') + const unknown = render(reasoning({ state: undefined, completedAt: undefined })) + expect(textIn(unknown.root)).toContain('Reasoning') + // No "Reasoning: Reasoning". + expect(toggleOf(unknown).props.accessibilityLabel).toBe('Reasoning') + expect(textIn(render(reasoning({ completedAt: 1_300 })).root)).toContain('Thought for 1s') + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatMessage.tsx b/mobile/src/session/MobileNativeChatMessage.tsx index 83f93266d6f..ec0a6d48b31 100644 --- a/mobile/src/session/MobileNativeChatMessage.tsx +++ b/mobile/src/session/MobileNativeChatMessage.tsx @@ -1,8 +1,9 @@ import { MobileSelectableText as Text } from '../components/MobileSelectableText' -import { memo, useCallback, useState, type ComponentProps, type ReactNode } from 'react' -import { Image, Text as NativeText, Pressable, View } from 'react-native' +import { memo, useCallback, useState } from 'react' +import { Image, Text as NativeText, View } from 'react-native' import { INLINE_TEXT_SELECTION } from '../components/inline-text-selection' import { MobileNativeChatMessageActionsSheet } from './MobileNativeChatMessageActionsSheet' +import { MobileNativeChatLongPressContent as Content } from './MobileNativeChatLongPressContent' import { splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' import { selectActiveToolCall } from '../../../src/shared/native-chat-tool-activity' import { isImageRefBlock, isTextBlock } from '../../../src/shared/native-chat-types' @@ -12,6 +13,8 @@ import { } from '../../../src/shared/agent-session-host-status-rows' import type { NativeChatBlock, NativeChatMessage } from '../../../src/shared/native-chat-types' import { MobileMarkdown } from '../components/MobileMarkdown' +import { deriveNativeChatRowContent } from '../../../src/shared/native-chat-row-content' +import { MobileNativeChatReasoningRow } from './MobileNativeChatReasoningRow' import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' import { ToolRun } from './MobileNativeChatToolRun' import type { NativeChatTurnStatus } from './use-mobile-native-chat-turn-status' @@ -88,25 +91,6 @@ function Prose({ return null } -// Keep the existing responder hierarchy on platforms with inline selection. -function Content({ - onLongPress, - style, - children -}: { - onLongPress?: () => void - style: ComponentProps['style'] - children: ReactNode -}): React.JSX.Element { - return onLongPress ? ( - - {children} - - ) : ( - {children} - ) -} - function MobileNativeChatMessageImpl({ message, toolsExpanded = false, @@ -118,7 +102,10 @@ function MobileNativeChatMessageImpl({ turnKey, onToggleTurn, activeTurnIsWorking, - structuredActivityUi = false + structuredActivityUi = false, + reasoningIsLive = false, + reasoningExpanded, + onToggleReasoning }: { message: NativeChatMessage toolsExpanded?: boolean @@ -139,6 +126,11 @@ function MobileNativeChatMessageImpl({ activeTurnIsWorking?: boolean /** Structured lane only: live tool progress plus the turn-status disclosure. */ structuredActivityUi?: boolean + /** This open reasoning block is disclosed by the live activity line, so its row draws nothing. */ + reasoningIsLive?: boolean + /** The transcript-held disclosure of a reasoning row; one stable handler takes its key. */ + reasoningExpanded?: boolean + onToggleReasoning?: (key: string) => void }): React.JSX.Element { const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' @@ -175,15 +167,44 @@ function MobileNativeChatMessageImpl({ onToggleExpanded={turnKey && onToggleTurn ? () => onToggleTurn(turnKey) : undefined} /> ) : null + if (isReasoning) { + // The same text the live line's selector reads, so the two agree on whether there is any. + const markdown = deriveNativeChatRowContent(message.blocks).markdown + // Blank, or disclosed by the live activity line: nothing draws, not even an empty row. + const draws = markdown.trim().length > 0 && !reasoningIsLive + return ( + <> + {turnStatusAbove ? statusRow : null} + {draws ? ( + + + + ) : null} + {actionsOpen ? ( + setActionsOpen(false)} + /> + ) : null} + {turnStatusAbove ? null : statusRow} + + ) + } return ( <> {/* A turn with no user bubble carries its bar above its first row. */} {turnStatusAbove ? statusRow : null} - + {prose.map((block, index) => ( + markdown: string + fontScale: number + /** Drawn inside its working turn: an open row there has not ended. */ + live?: boolean + /** The transcript's disclosure, keyed like the live line's; absent, the row keeps its own. */ + expanded?: boolean + onToggle?: (key: string) => void + onOpenFile?: (relativePath: string) => void + /** Android only: opens the message's actions sheet, as a long press on any other message does. */ + onLongPress?: () => void +}): React.JSX.Element { + const [localExpanded, setLocalExpanded] = useState(false) + const expanded = onToggle ? transcriptExpanded : localExpanded + const toggle = () => + onToggle ? onToggle(nativeChatReasoningDisclosureKey(message.id)) : setLocalExpanded(!expanded) + const headline = nativeChatReasoningHeadlineText(nativeChatReasoningHeadline(message, { live })) + const label = nativeChatReasoningHeadlineText({ kind: 'reasoning' }) + return ( + + [styles.reasoningToggle, pressed && styles.reasoningPressed]} + onPress={toggle} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded }} + // Desktop's screen-reader prefix: the headline alone does not say what was thought. + accessibilityLabel={headline === label ? label : `${label}: ${headline}`} + > + + + {headline} + + + + + + {expanded ? ( + + ) : null} + + ) +} + +/** A reasoning block's text, under its row or the live activity line. Capped and scrollable, so an + * open block streaming at the tail cannot grow without bound. */ +export function MobileNativeChatReasoningBody({ + markdown, + fontScale, + onOpenFile, + onLongPress +}: { + markdown: string + fontScale: number + onOpenFile?: (relativePath: string) => void + onLongPress?: () => void +}): React.JSX.Element { + return ( + + + + + + ) +} diff --git a/mobile/src/session/MobileNativeChatTurnStatus.test.ts b/mobile/src/session/MobileNativeChatTurnStatus.test.ts index a0393996cec..2506ab6e311 100644 --- a/mobile/src/session/MobileNativeChatTurnStatus.test.ts +++ b/mobile/src/session/MobileNativeChatTurnStatus.test.ts @@ -19,10 +19,7 @@ vi.mock('react-native', async () => { }) vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' })) -import { - MobileNativeChatTurnActivity, - MobileNativeChatTurnStatus -} from './MobileNativeChatTurnStatus' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' const labels = (node: ReactTestInstance): string[] => node.findAllByType('Text' as never).map((text) => String(text.children.join(''))) @@ -108,38 +105,3 @@ describe('MobileNativeChatTurnStatus', () => { expect(vi.getTimerCount()).toBe(0) }) }) - -describe('MobileNativeChatTurnActivity', () => { - function render(props: { thinking: boolean; activityText?: string | null }): ReactTestRenderer { - act(() => { - renderer = create(createElement(MobileNativeChatTurnActivity, props)) - }) - return renderer! - } - - it('reads "Thinking" beside one spinner while the turn reasons', () => { - const tree = render({ thinking: true }) - expect(labels(tree.root)).toEqual(['Thinking']) - expect(spinners(tree.root)).toHaveLength(1) - }) - - // The bar owns the clock; the tail line never repeats it. - it('reads plain "Working…" when the turn is not reasoning, and holds no timer', () => { - const tree = render({ thinking: false }) - expect(labels(tree.root)).toEqual(['Working…']) - expect(vi.getTimerCount()).toBe(0) - }) - - it('lets provider activity text beat both fallbacks', () => { - const tree = render({ thinking: true, activityText: 'Running pnpm test' }) - expect(labels(tree.root)).toEqual(['Running pnpm test']) - expect(spinners(tree.root)).toHaveLength(1) - }) - - it('announces the live line to assistive tech', () => { - const tree = render({ thinking: true }) - const row = tree.root.findByType('View' as never) - expect(row.props.accessibilityLiveRegion).toBe('polite') - expect(row.props.accessibilityLabel).toBe('Agent is responding') - }) -}) diff --git a/mobile/src/session/MobileNativeChatTurnStatus.tsx b/mobile/src/session/MobileNativeChatTurnStatus.tsx index a39e404c639..9bb6a5c9e0f 100644 --- a/mobile/src/session/MobileNativeChatTurnStatus.tsx +++ b/mobile/src/session/MobileNativeChatTurnStatus.tsx @@ -1,8 +1,7 @@ import { useEffect, useState } from 'react' -import { ActivityIndicator, Pressable, StyleSheet, Text, View } from 'react-native' +import { Pressable, StyleSheet, Text, View } from 'react-native' import { ChevronRight } from 'lucide-react-native' import { - formatNativeChatActiveTurnLabel, formatNativeChatTurnStatusLabel, NATIVE_CHAT_TURN_STATUS_COPY, nativeChatElapsedSeconds @@ -76,30 +75,6 @@ export function MobileNativeChatTurnStatus({ ) } -/** The live turn's tail line: a spinner beside what the provider says it is doing, - * else "Thinking", else "Working…". The clock stays in the turn bar. Desktop - * parity: `NativeChatTurnActivityLine`. */ -export function MobileNativeChatTurnActivity({ - thinking, - activityText -}: { - thinking: boolean - activityText?: string | null -}): React.JSX.Element { - return ( - - - - {formatNativeChatActiveTurnLabel({ activityText, thinking })} - - - ) -} - const styles = StyleSheet.create({ row: { flexDirection: 'row', diff --git a/mobile/src/session/MobileNativeChatView.test.ts b/mobile/src/session/MobileNativeChatView.test.ts index ff2637f8a45..d96c9179b08 100644 --- a/mobile/src/session/MobileNativeChatView.test.ts +++ b/mobile/src/session/MobileNativeChatView.test.ts @@ -47,6 +47,7 @@ vi.mock('lucide-react-native', () => ({ })) vi.mock('./MobileNativeChatMessage', () => ({ MobileNativeChatMessage: 'ChatMessage' })) +vi.mock('./MobileNativeChatLiveLine', () => ({ MobileNativeChatLiveLine: 'LiveStatus' })) vi.mock('./MobileNativeChatAsk', () => ({ MobileNativeChatAsk: 'ChatAsk' })) vi.mock('./MobileNativeChatPermission', () => ({ MobileNativeChatPermission: 'ChatPermission' })) vi.mock('./MobileNativeChatQuestion', () => ({ MobileNativeChatQuestion: 'ChatQuestion' })) diff --git a/mobile/src/session/MobileNativeChatView.tsx b/mobile/src/session/MobileNativeChatView.tsx index 8e729e5e46b..a559bd02ef2 100644 --- a/mobile/src/session/MobileNativeChatView.tsx +++ b/mobile/src/session/MobileNativeChatView.tsx @@ -30,7 +30,7 @@ import { useMobileNativeChatPinchGesture } from './use-mobile-native-chat-pinch- import { useMobileNativeChatTailFollow } from './use-mobile-native-chat-tail-follow' import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' import { useSettledMobileNativeChatInputLock } from './use-mobile-native-chat-input-lease' -import { MobileNativeChatTurnActivity } from './MobileNativeChatTurnStatus' +import { MobileNativeChatLiveLine } from './MobileNativeChatLiveLine' import { MobileAgentWorkingIndicator } from './MobileAgentWorkingIndicator' import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment' import { MobileNativeChatComposer } from './MobileNativeChatComposer' @@ -278,10 +278,9 @@ export function MobileNativeChatView({ turnJournal, thinking: turnIndicator?.thinking === true, activityText: turnIndicator?.activityText ?? null, + lineYields: structuredActivityUi && (ask != null || permission != null || question != null), scopeKey: sendSurfaceId }) - const hasPendingStructuredInteraction = - structuredActivityUi && (ask != null || permission != null || question != null) const renderItem = useCallback( ({ item, index }: { item: NativeChatMessage; index: number }) => ( @@ -298,13 +297,14 @@ export function MobileNativeChatView({ [toolsExpanded, fontScale, onOpenFile, structuredActivityUi, turns] ) - const liveStatus = - structuredActivityUi && agentWorking && !hasPendingStructuredInteraction && turns.active ? ( - - ) : null + const liveStatus = turns.liveLine ? ( + + ) : null const emptyState = mobileNativeChatEmptyState(status, agent ?? null, error) const showLoading = status === 'loading' && messages.length === 0 diff --git a/mobile/src/session/MobileNativeChatView.turn-status.test.ts b/mobile/src/session/MobileNativeChatView.turn-status.test.ts index 1f8191df80b..47137a7deaa 100644 --- a/mobile/src/session/MobileNativeChatView.turn-status.test.ts +++ b/mobile/src/session/MobileNativeChatView.turn-status.test.ts @@ -47,6 +47,7 @@ vi.mock('lucide-react-native', () => ({ })) vi.mock('./MobileNativeChatMessage', () => ({ MobileNativeChatMessage: 'ChatMessage' })) +vi.mock('./MobileNativeChatLiveLine', () => ({ MobileNativeChatLiveLine: 'LiveStatus' })) vi.mock('./MobileNativeChatAsk', () => ({ MobileNativeChatAsk: 'ChatAsk' })) vi.mock('./MobileNativeChatPermission', () => ({ MobileNativeChatPermission: 'ChatPermission' })) vi.mock('./MobileNativeChatQuestion', () => ({ MobileNativeChatQuestion: 'ChatQuestion' })) @@ -177,13 +178,11 @@ describe('MobileNativeChatView', () => { return (renderedRow(id) as { props: Record }).props } + /** What the live footer line says: its label inputs, read off the line the view hands it. */ function footerProps(): Record | null { const list = renderer!.root.find((node) => String(node.type) === 'FlatList') - const footer = list.props.ListFooterComponent as - | { props: Record } - | null - | undefined - return footer?.props ?? null + const line = list.props.ListFooterComponent?.props.line + return line ? { thinking: line.thinking, activityText: line.activityText } : null } function workingIndicators(): ReactTestInstance[] { diff --git a/mobile/src/session/MobileNativeChatView.waiting-rows.test.ts b/mobile/src/session/MobileNativeChatView.waiting-rows.test.ts index 518d85f2465..f4916c0a324 100644 --- a/mobile/src/session/MobileNativeChatView.waiting-rows.test.ts +++ b/mobile/src/session/MobileNativeChatView.waiting-rows.test.ts @@ -37,7 +37,7 @@ vi.mock('lucide-react-native', () => ({ Square: 'Square' })) vi.mock('./MobileNativeChatMessage', () => ({ MobileNativeChatMessage: 'ChatMessage' })) -vi.mock('./MobileNativeChatTurnStatus', () => ({ MobileNativeChatTurnActivity: 'LiveStatus' })) +vi.mock('./MobileNativeChatLiveLine', () => ({ MobileNativeChatLiveLine: 'LiveStatus' })) vi.mock('./MobileNativeChatComposer', () => ({ MobileNativeChatComposer: 'Composer' })) // The queue's action sheet pulls in the animation runtime, which this react-native mock can't host. vi.mock('../components/ActionSheetModal', () => ({ ActionSheetModal: 'ActionSheetModal' })) diff --git a/mobile/src/session/mobile-native-chat-message-styles.ts b/mobile/src/session/mobile-native-chat-message-styles.ts index 225d664e65e..2acad402176 100644 --- a/mobile/src/session/mobile-native-chat-message-styles.ts +++ b/mobile/src/session/mobile-native-chat-message-styles.ts @@ -35,7 +35,31 @@ export const styles = StyleSheet.create({ lineHeight: TEXT_SIZE + 6 }, reasoning: { - opacity: 0.7 + opacity: 0.7, + // Starts under the headline, past the 15 pt brain and its gap. + paddingLeft: 15 + spacing.sm + }, + reasoningToggle: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm, + // With the toggle's 6 pt hitSlop above and below, a 44 pt touch target. + minHeight: 32 + }, + reasoningPressed: { + opacity: 0.6 + }, + reasoningHeadline: { + color: colors.textMuted, + fontSize: typography.bodySize, + flexShrink: 1 + }, + reasoningCaretOpen: { + transform: [{ rotate: '90deg' }] + }, + reasoningBody: { + // The common cap for an open reasoning block (about ten lines). + maxHeight: 240 }, toolRun: { marginTop: spacing.xs diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx index 19f0dfbc0db..5a29875b08c 100644 --- a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx @@ -29,6 +29,8 @@ function Harness({ settledTurns, turnJournal, workingStartedAt, + thinking, + lineYields, scopeKey = 'host\0worktree\0tab-a' }: { messages: readonly NativeChatMessage[] @@ -37,6 +39,8 @@ function Harness({ settledTurns?: NativeChatSettledTurns turnJournal?: NativeChatTurnJournal workingStartedAt?: number | null + thinking?: boolean + lineYields?: boolean scopeKey?: string }): React.JSX.Element { const disclosure = useMobileNativeChatTurnDisclosure({ @@ -46,6 +50,8 @@ function Harness({ settledTurns, turnJournal, workingStartedAt, + thinking, + lineYields, scopeKey }) return createElement('result', { disclosure }) @@ -762,3 +768,65 @@ describe('useMobileNativeChatTurnDisclosure', () => { expect(rows[0].turnStatus?.workedSeconds).toBe(4) }) }) + +describe('the open reasoning block the live line discloses', () => { + let renderer: ReactTestRenderer | null = null + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + }) + + const block = (id: string, state: 'running' | 'completed'): NativeChatMessage => ({ + id, + role: 'reasoning', + blocks: [{ type: 'text', text: `${id} weighs two approaches` }], + timestamp: null, + source: 'transcript', + state + }) + const show = ( + messages: NativeChatMessage[], + props: { thinking?: boolean; lineYields?: boolean } + ) => + act(() => { + const element = createElement(Harness, { messages, enabled: true, ...props }) + if (renderer) { + renderer.update(element) + } else { + renderer = create(element) + } + }) + const latest = () => renderer!.root.findByType('result').props.disclosure + + it('hides only that block, and lands it open once it ends if the reader opened it live', () => { + const prompt = userMessage('u1') + show([prompt, block('r-1', 'running')], { thinking: true }) + expect(latest().liveLine).toMatchObject({ + reasoning: { message: { id: 'r-1' } }, + reasoningExpanded: false + }) + expect(latest().resolveRow(1, block('r-1', 'running')).reasoningIsLive).toBe(true) + act(() => latest().onToggleReasoning('reasoning:r-1')) + expect(latest().liveLine.reasoningExpanded).toBe(true) + + // Closed while the turn works on: the line discloses nothing, and the row draws open. + show([prompt, block('r-1', 'completed')], { thinking: false }) + expect(latest().liveLine).toMatchObject({ reasoning: null }) + const row = latest().resolveRow(1, block('r-1', 'completed')) + expect(row).toMatchObject({ reasoningIsLive: false, reasoningExpanded: true }) + + // The next block starts collapsed. + show([prompt, block('r-1', 'completed'), block('r-2', 'running')], { thinking: true }) + expect(latest().liveLine).toMatchObject({ + reasoning: { message: { id: 'r-2' } }, + reasoningExpanded: false + }) + }) + + it('discloses nothing, and hides nothing, while a prompt takes the line', () => { + show([userMessage('u1'), block('r-1', 'running')], { thinking: true, lineYields: true }) + expect(latest().liveLine).toBeNull() + expect(latest().resolveRow(1, block('r-1', 'running')).reasoningIsLive).toBe(false) + }) +}) diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts index 6e1d4df395e..4c7262efc70 100644 --- a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts @@ -1,7 +1,13 @@ import { useCallback, useMemo, useState } from 'react' import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { nativeChatReasoningDisclosureKey } from '../../../src/shared/native-chat-reasoning-row' +import { + nativeChatLiveLine, + type NativeChatLiveLine +} from '../../../src/shared/native-chat-live-line' import type { NativeChatSettledTurns } from '../../../src/shared/native-chat-turn-status' import { + isNativeChatRowInLiveWorkingTurn, nativeChatMessagesWaitingBehindLiveTurn, nativeChatTurnMembership, type NativeChatTurnJournal @@ -27,6 +33,41 @@ export type MobileNativeChatTurnRow = { /** Set only on a settled turn — the one row that has activity to disclose. */ turnKey?: string activeTurnIsWorking: boolean + /** The live activity line discloses this open reasoning block, so its row draws nothing. */ + reasoningIsLive: boolean + /** A reasoning row's disclosure, keyed like the live line's so an opened block stays open. */ + reasoningExpanded: boolean + onToggleReasoning: (key: string) => void +} + +/** The live activity line, when it draws, and whether the reader opened the block it discloses. */ +export type MobileNativeChatLiveLine = NativeChatLiveLine & { reasoningExpanded: boolean } + +/** Keys the reader opened in this chat, bounded; another chat's never leak in. */ +function useScopedOpenKeys(scopeKey: string): [ReadonlySet, (key: string) => void] { + const [state, setState] = useState<{ scopeKey: string; keys: ReadonlySet }>(() => ({ + scopeKey, + keys: new Set() + })) + const toggle = useCallback( + (key: string) => { + setState((current) => { + const next = new Set(current.scopeKey === scopeKey ? current.keys : []) + if (!next.delete(key)) { + if (next.size >= MAX_EXPANDED_TURNS) { + const oldest = next.values().next().value + if (oldest) { + next.delete(oldest) + } + } + next.add(key) + } + return { scopeKey, keys: next } + }) + }, + [scopeKey] + ) + return [state.scopeKey === scopeKey ? state.keys : EMPTY_TURN_IDS, toggle] } /** Owns the transcript's per-turn status rows and their disclosure state, and @@ -41,6 +82,7 @@ export function useMobileNativeChatTurnDisclosure({ turnJournal = null, thinking = false, activityText = null, + lineYields = false, scopeKey }: { messages: readonly NativeChatMessage[] @@ -55,6 +97,8 @@ export function useMobileNativeChatTurnDisclosure({ thinking?: boolean /** What the provider says the live turn is doing; outranks the other labels. */ activityText?: string | null + /** A prompt the reader must answer replaces the live activity line. */ + lineYields?: boolean /** Host/worktree/tab identity for timing and disclosure isolation. */ scopeKey: string }): { @@ -67,6 +111,10 @@ export function useMobileNativeChatTurnDisclosure({ listMessages: readonly NativeChatMessage[] /** Rows waiting behind the live turn, drawn after its live status. */ waitingRows: readonly { item: NativeChatMessage; index: number }[] + /** The live activity line, or null while the turn is idle or a prompt replaces it. */ + liveLine: MobileNativeChatLiveLine | null + /** Stable for a given chat scope, so it never disturbs a row's memo. */ + onToggleReasoning: (key: string) => void } { // Resolve each row's turn, which turn is live, and the order the rows draw in, once: from the // turn record when the host states scopes, else by journal order. @@ -103,34 +151,39 @@ export function useMobileNativeChatTurnDisclosure({ thinking, scopeKey }) - const [expandedTurns, setExpandedTurns] = useState<{ - scopeKey: string - turnIds: ReadonlySet - }>(() => ({ scopeKey, turnIds: new Set() })) - const expandedTurnIds = - expandedTurns.scopeKey === scopeKey ? expandedTurns.turnIds : EMPTY_TURN_IDS - const toggleExpandedTurn = useCallback( - (turnKey: string) => { - setExpandedTurns((current) => { - const next = new Set(current.scopeKey === scopeKey ? current.turnIds : []) - if (!next.delete(turnKey)) { - if (next.size >= MAX_EXPANDED_TURNS) { - const oldest = next.values().next().value - if (oldest) { - next.delete(oldest) - } - } - next.add(turnKey) - } - return { scopeKey, turnIds: next } - }) - }, - [scopeKey] - ) + const [expandedTurnIds, toggleExpandedTurn] = useScopedOpenKeys(scopeKey) + const [expandedReasoning, toggleReasoning] = useScopedOpenKeys(scopeKey) const bars = useMemo(() => nativeChatTurnBarRows(rows, turnKeys), [rows, turnKeys]) const { active, activeTurnKey, completedByTurn } = turnStatuses + const inLiveWorkingTurn = useCallback( + (index: number) => + isNativeChatRowInLiveWorkingTurn(turnKeys[index], liveTurnKey, enabled && isWorking), + [enabled, isWorking, liveTurnKey, turnKeys] + ) const activeActivityText = enabled && isWorking ? (activityText ?? null) : null + const line = useMemo( + () => + nativeChatLiveLine({ + draws: enabled && isWorking && !lineYields && active !== null, + thinking: active?.thinking === true, + activityText: activeActivityText, + messages: rows, + inLiveWorkingTurn + }), + [active, activeActivityText, enabled, inLiveWorkingTurn, isWorking, lineYields, rows] + ) + const liveReasoningId = line?.reasoning?.message.id + const liveLine = useMemo( + () => + line && { + ...line, + reasoningExpanded: + line.reasoning !== null && + expandedReasoning.has(nativeChatReasoningDisclosureKey(line.reasoning.message.id)) + }, + [expandedReasoning, line] + ) const resolveRow = useCallback( (listIndex: number, message: NativeChatMessage): MobileNativeChatTurnRow => { const index = waiting.indexById?.get(message.id) ?? listIndex @@ -153,23 +206,28 @@ export function useMobileNativeChatTurnDisclosure({ // transcript, defeating the row's memo; caching one per turn would mean // writing a ref during render, which react-freeze can discard. turnKey: turnKey && turnStatus?.workedSeconds != null ? turnKey : undefined, - // Liveness is the live turn's rows, not the newest prompt's: a running turn's rows stay live - // while a newer message waits behind it. With no user boundary at all, the session's - // working state stays authoritative. - activeTurnIsWorking: enabled && isWorking && turnKey === liveTurnKey + // With no user boundary at all, the session's working state stays authoritative. + activeTurnIsWorking: inLiveWorkingTurn(index), + reasoningIsLive: message.id === liveReasoningId, + reasoningExpanded: + message.role === 'reasoning' && + expandedReasoning.has(nativeChatReasoningDisclosureKey(message.id)), + onToggleReasoning: toggleReasoning } }, [ turnKeys, waiting, bars, - liveTurnKey, enabled, activeTurnKey, active, completedByTurn, expandedTurnIds, - isWorking + inLiveWorkingTurn, + liveReasoningId, + expandedReasoning, + toggleReasoning ] ) @@ -180,6 +238,8 @@ export function useMobileNativeChatTurnDisclosure({ onToggleTurn: toggleExpandedTurn, resolveRow, listMessages: waiting.listMessages, - waitingRows: waiting.waitingRows + waitingRows: waiting.waitingRows, + liveLine, + onToggleReasoning: toggleReasoning } } diff --git a/src/main/claude/claude-hook-event-versions.ts b/src/main/claude/claude-hook-event-versions.ts index 5c9c28de66c..2e980cd0477 100644 --- a/src/main/claude/claude-hook-event-versions.ts +++ b/src/main/claude/claude-hook-event-versions.ts @@ -60,7 +60,12 @@ export function claudeKnowsStatusLine(version: string | null | undefined): boole return claudeKnowsSince(version, CLAUDE_STATUS_LINE_FIRST_VERSION) } -export async function probeClaudeCliVersion(executablePath: string): Promise { +/** `launch` runs the probe with the cwd and env a launch will spawn the CLI with, so a version + * manager's shim picks the same CLI the launch will. */ +export async function probeClaudeCliVersion( + executablePath: string, + launch?: { cwd: string; env: Record; timeoutMs?: number } +): Promise { try { const pathKey = process.platform === 'win32' && process.env.Path !== undefined ? 'Path' : 'PATH' const executableDir = path.dirname(executablePath) @@ -70,13 +75,14 @@ export async function probeClaudeCliVersion(executablePath: string): Promise streamedBlocks: ReturnType streamedText: ReturnType + streamedThinking: ClaudeStreamedThinking subagents: ClaudeSubagentRoster toolOrigins: ClaudeToolOriginRegistry backgroundTasks: ClaudeBackgroundTaskRows @@ -97,10 +92,11 @@ export function journalClaudeMessage( const outputEnvelope = claudeOutputEnvelope(envelope) const body = claudeMessageBody(outputEnvelope) const identity = - (body && envelope.role === 'assistant' ? ctx.streamedBlocks.reconcile(envelope) : null) ?? - claudeMessageIdentity(envelope) + (body && envelope.role === 'assistant' + ? ctx.streamedBlocks.reconcile(envelope)?.identity + : null) ?? claudeMessageIdentity(envelope) ctx.streamedText.forget(agentJournalItemKey(identity)) - const thinking = claudeThinkingText(outputEnvelope) + const thinking = ctx.streamedThinking.finalize(outputEnvelope, observedAt) const source: ClaudeTurnSource = { sessionId: envelope.sessionId, uuid: envelope.uuid, @@ -155,15 +151,12 @@ export function journalClaudeMessage( } if (thinking) { ctx.turn.ensureOpen(message, source, observedAt) - const thinkingIdentity = claudeThinkingIdentity(envelope.sessionId, envelope.uuid) - const thinkingBody: AgentJournalItemBody = { - kind: 'message', - role: 'reasoning', - blocks: [ - { type: 'text', text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text } - ] - } - ctx.sink.appendItem(thinkingIdentity, thinkingBody, stamp(thinkingIdentity, thinkingBody)) + // The write that ends the row: shed under pressure, the row would read open for good. + ctx.sink.appendItem(thinking.identity, thinking.body, { + ...stamp(thinking.identity, thinking.body), + ...(thinking.startedAt === undefined ? {} : { observedAt: thinking.startedAt }), + lifecycle: true + }) changed = true } changed = diff --git a/src/main/claude/claude-open-turn.ts b/src/main/claude/claude-open-turn.ts index 32149bf21f3..84b4ed44c68 100644 --- a/src/main/claude/claude-open-turn.ts +++ b/src/main/claude/claude-open-turn.ts @@ -26,6 +26,8 @@ export type ClaudeOpenTurnDeps = { sink: StructuredAgentSessionEventSink /** Settles the superseded turn's children; they get no later event of their own. */ settleChildren: (groupKey: string | null) => void + /** Ends what the ending turn left open, at the instant it ended; no later frame will. */ + endOpenWork: (completedAt: number) => void /** A turn opening moves the conversation on. */ onOpen?: () => void } @@ -111,6 +113,7 @@ export class ClaudeOpenTurn { this.deps.onOpen?.() if (this.current) { this.deps.settleChildren(this.groupKey) + this.deps.endOpenWork(observedAt) this.publish(this.current, { state: 'interrupted', completedAt: observedAt, @@ -128,6 +131,7 @@ export class ClaudeOpenTurn { this.deps.onOpen?.() if (this.current) { this.deps.settleChildren(this.groupKey) + this.deps.endOpenWork(turn.startedAt) this.publish(this.current, { state: 'interrupted', completedAt: turn.startedAt, @@ -168,6 +172,8 @@ export class ClaudeOpenTurn { settle(end: ClaudeTurnEnd, contextUsage?: AgentSessionContextUsage): void { // Every settle is a provider cycle ending (result, idle, child exit). this.cycleWorkObserved = false + // Even with no turn open: work a suppressed turn produced still ends here. + this.deps.endOpenWork(end.completedAt) if (this.current) { this.publish(this.current, end, contextUsage) this.current = null diff --git a/src/main/claude/claude-streamed-block-identity.ts b/src/main/claude/claude-streamed-block-identity.ts index 00bfdd103ab..3ac025c1b08 100644 --- a/src/main/claude/claude-streamed-block-identity.ts +++ b/src/main/claude/claude-streamed-block-identity.ts @@ -14,24 +14,29 @@ export type ClaudeStreamedTextDelta = { * prose has no message envelope when it is persisted, so the producer travels * with the delta rather than being re-read from a frame that is long gone. */ parentToolUseId: string | null + /** Host clock when the block began: its `content_block_start`, else its first delta. Text can + * trail the start by seconds, so this, not the first write, is when the block started. */ + startedAt: number } +export type StreamedBlock = { identity: AgentJournalItemIdentity; startedAt: number } + type StreamedMessage = { messageId: string | null - blocks: Map + blocks: Map /** Streamed text blocks whose final assistant frame has not arrived, in block order. */ - awaitingFinal: AgentJournalItemIdentity[] + awaitingFinal: StreamedBlock[] } export type ClaudeStreamedBlockRegistry = { /** Text a stream_event frame appends to its block, or null when it carries none. */ - observe: (frame: Record) => ClaudeStreamedTextDelta | null - /** The streamed identity a final assistant frame reconciles onto, if its block streamed. */ + observe: (frame: Record, observedAt: number) => ClaudeStreamedTextDelta | null + /** The streamed block a final assistant frame reconciles onto, if its block streamed. */ reconcile: (frame: { sessionId: string parentToolUseId: string | null messageId: string | null - }) => AgentJournalItemIdentity | null + }) => StreamedBlock | null clear: () => void } @@ -39,7 +44,9 @@ function scopeKey(sessionId: string, parentToolUseId: string | null): string { return `${sessionId}/${parentToolUseId ?? ''}` } -export function createClaudeStreamedBlockRegistry(): ClaudeStreamedBlockRegistry { +export function createClaudeStreamedBlockRegistry( + blockType: 'text' | 'thinking' = 'text' +): ClaudeStreamedBlockRegistry { const messages = new Map() const messageFor = (scope: string): StreamedMessage => { @@ -55,16 +62,18 @@ export function createClaudeStreamedBlockRegistry(): ClaudeStreamedBlockRegistry streamed: StreamedMessage, sessionId: string, index: number, - uuid: string - ): AgentJournalItemIdentity => { + uuid: string, + startedAt: number + ): StreamedBlock => { const identity: AgentJournalItemIdentity = { provider: 'claude', sessionId, uuid } - streamed.blocks.set(index, identity) - streamed.awaitingFinal.push(identity) - return identity + const block = { identity, startedAt } + streamed.blocks.set(index, block) + streamed.awaitingFinal.push(block) + return block } return { - observe: (frame) => { + observe: (frame, observedAt) => { const event = claudeRecord(frame.event) const sessionId = claudeText(frame.session_id) const uuid = claudeText(frame.uuid) @@ -84,24 +93,25 @@ export function createClaudeStreamedBlockRegistry(): ClaudeStreamedBlockRegistry const index = typeof event.index === 'number' ? event.index : 0 if (event.type === 'content_block_start') { const block = claudeRecord(event.content_block) - if (block?.type !== 'text') { + if (block?.type !== blockType) { return null } - const identity = mint(messageFor(scope), sessionId, index, uuid) - const text = claudeText(block.text) - return text ? { identity, text, parentToolUseId } : null + const started = mint(messageFor(scope), sessionId, index, uuid, observedAt) + const text = claudeText(block[blockType]) + return text ? { ...started, text, parentToolUseId } : null } if (event.type !== 'content_block_delta') { return null } const delta = claudeRecord(event.delta) - const text = delta?.type === 'text_delta' ? claudeText(delta.text) : null + const text = delta?.type === `${blockType}_delta` ? claudeText(delta[blockType]) : null if (!text) { return null } const streamed = messageFor(scope) - const identity = streamed.blocks.get(index) ?? mint(streamed, sessionId, index, uuid) - return { identity, text, parentToolUseId } + const started = + streamed.blocks.get(index) ?? mint(streamed, sessionId, index, uuid, observedAt) + return { ...started, text, parentToolUseId } }, reconcile: (frame) => { const streamed = messages.get(scopeKey(frame.sessionId, frame.parentToolUseId)) diff --git a/src/main/claude/claude-streamed-text-checkpoints.ts b/src/main/claude/claude-streamed-text-checkpoints.ts index ad21ee790a6..cf8d3c40be5 100644 --- a/src/main/claude/claude-streamed-text-checkpoints.ts +++ b/src/main/claude/claude-streamed-text-checkpoints.ts @@ -9,11 +9,13 @@ import { import type { ClaudeSubagentLinkageSource } from './claude-subagent-linkage' export type ClaudeStreamedTextCheckpointDeps = { - /** Rewrites the block's journal row with the text accumulated so far. */ + /** Rewrites the block's journal row with the text accumulated so far. `ended` is set on the + * block's last write, when it ends without the final frame that would otherwise replace it. */ persist: ( identity: AgentJournalItemIdentity, text: string, - options: StructuredAgentSessionAppendOptions + options: StructuredAgentSessionAppendOptions, + ended?: ClaudeStreamedBlockEnd ) => void /** Who produced a block, asked by the scope the block streamed under. */ producer: ClaudeSubagentLinkageSource @@ -21,6 +23,9 @@ export type ClaudeStreamedTextCheckpointDeps = { schedule?: AgentSessionDeltaCoalescerDeps['schedule'] } +/** How a block ended: `completedAt` is when the host saw it end, or saw what cut it off. */ +export type ClaudeStreamedBlockEnd = { completedAt?: number } + export type ClaudeStreamedTextCheckpoints = { /** Accumulate a delta; the row is rewritten on the coalescer's own cadence. */ append: ( @@ -36,6 +41,11 @@ export type ClaudeStreamedTextCheckpoints = { reattribute: () => void /** Drop one block's state, for a block whose final frame has now landed. */ forget: (key: string) => void + /** The text received for a block so far, if it is still streaming. */ + latest: (key: string) => string | undefined + /** Write each matching block's text one last time as ended, then drop it: its final frame is + * never coming. A block with no text has no row and is only dropped. */ + finish: (ended: ClaudeStreamedBlockEnd, only?: (key: string) => boolean) => void /** * Drop every block still awaiting its final frame, at turn settlement. Their * text is already journaled by the flush that precedes settlement; keeping it @@ -156,6 +166,20 @@ export function createClaudeStreamedTextCheckpoints( } }, forget: drop, + latest: (key) => coalescer.snapshot(key)?.text ?? latestText.get(key), + finish: (ended, only) => { + for (const [key, identity] of identities) { + if (only && !only(key)) { + continue + } + coalescer.flush(key) + const text = latestText.get(key) + if (text !== undefined) { + deps.persist(identity, text, producerOptions(key), ended) + } + drop(key) + } + }, settle: () => { // Map iteration tolerates deletion of the entry just visited. for (const key of identities.keys()) { diff --git a/src/main/claude/claude-streamed-thinking.ts b/src/main/claude/claude-streamed-thinking.ts new file mode 100644 index 00000000000..1edd109adeb --- /dev/null +++ b/src/main/claude/claude-streamed-thinking.ts @@ -0,0 +1,156 @@ +import type { + AgentJournalItemIdentity, + AgentJournalMessageItem, + AgentJournalTurnScope +} from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { + endedJournalReasoning, + journalReasoningBody +} from '../native-chat/agent-session-journal/journal-reasoning-row' +import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' +import { + createClaudeStreamedTextCheckpoints, + type ClaudeStreamedBlockEnd +} from './claude-streamed-text-checkpoints' +import { + claudeRecord, + claudeText, + claudeThinkingIdentity, + claudeThinkingText, + type ClaudeMessageEnvelope +} from './claude-structured-item-translation' +import type { ClaudeSubagentLinkageSource } from './claude-subagent-linkage' + +/** The stream a frame belongs to, which a new message in it starts over. */ +function streamScope(frame: Record): string | null { + const event = claudeRecord(frame.event) + const sessionId = claudeText(frame.session_id) + if (frame.type !== 'stream_event' || event?.type !== 'message_start' || !sessionId) { + return null + } + return `${sessionId}/${claudeText(frame.parent_tool_use_id) ?? ''}` +} + +export type ClaudeStreamedThinking = ReturnType + +/** A thinking block's closing row, and when the block began if it was seen streaming. */ +export type ClaudeReasoningFinal = { + identity: AgentJournalItemIdentity + body: AgentJournalMessageItem + startedAt?: number +} + +/** Thinking blocks streamed under --include-partial-messages, written to the same row their + * final assistant frame later lands on, and open until that frame or the turn's end. */ +export function createClaudeStreamedThinking(deps: { + sink: StructuredAgentSessionEventSink + producer: ClaudeSubagentLinkageSource + turnScope: () => AgentJournalTurnScope + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] +}) { + const blocks = createClaudeStreamedBlockRegistry('thinking') + /** Each open block's stream, and when it began. */ + const open = new Map() + const checkpoints = createClaudeStreamedTextCheckpoints({ + ...(deps.coalesceMs === undefined ? {} : { coalesceMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + producer: deps.producer, + persist: (identity, text, options, ended) => { + const body = journalReasoningBody( + text, + ended ? endedJournalReasoning(ended.completedAt) : { state: 'running' } + ) + if (body) { + const startedAt = open.get(agentJournalItemKey(identity))?.startedAt + deps.sink.appendItem(identity, body, { + ...options, + turnScope: deps.turnScope(), + // The row starts with its block, not with its first text or a queued write. + ...(startedAt === undefined ? {} : { observedAt: startedAt }), + // An end the sink sheds under pressure would leave the row open with nothing to close it. + ...(ended ? { lifecycle: true } : {}) + }) + deps.sink.publish() + } + } + }) + const finish = (ended: ClaudeStreamedBlockEnd, scope?: string): void => { + checkpoints.finish( + ended, + scope === undefined ? undefined : (key) => open.get(key)?.scope === scope + ) + for (const [key, block] of open) { + if (scope === undefined || block.scope === scope) { + open.delete(key) + } + } + } + + return { + /** True when the frame carried thinking text. */ + observe: (frame: Record, observedAt: number): boolean => { + // A new message in a stream means the previous one's unfinished blocks are never finishing. + const restarted = streamScope(frame) + if (restarted !== null) { + finish({ completedAt: observedAt }, restarted) + } + const delta = blocks.observe(frame, observedAt) + if (!delta) { + return false + } + const identity = delta.identity + if (identity.provider === 'claude') { + open.set(agentJournalItemKey(identity), { + scope: `${identity.sessionId}/${delta.parentToolUseId ?? ''}`, + startedAt: delta.startedAt + }) + } + checkpoints.append(identity, delta.text, delta.parentToolUseId) + return true + }, + /** The row a final frame's thinking lands on — its streamed block's, else its own — closed. + * Only a block seen streaming has an observed end. */ + finalize: ( + envelope: ClaudeMessageEnvelope, + observedAt: number + ): ClaudeReasoningFinal | null => { + if (!envelope.content.some((part) => claudeRecord(part)?.type === 'thinking')) { + return null + } + const streamed = blocks.reconcile(envelope) + const identity = + streamed?.identity ?? claudeThinkingIdentity(envelope.sessionId, envelope.uuid) + const key = agentJournalItemKey(identity) + // A final frame with no text of its own still ends the row its stream wrote. + const text = claudeThinkingText(envelope) ?? checkpoints.latest(key) ?? '' + checkpoints.forget(key) + open.delete(key) + const body = journalReasoningBody( + text, + endedJournalReasoning(streamed ? observedAt : undefined) + ) + return body + ? { identity, body, ...(streamed ? { startedAt: streamed.startedAt } : {}) } + : null + }, + /** End every block still open, for a turn that is ending. */ + finishOpen: (completedAt: number): void => { + finish({ completedAt }) + blocks.clear() + }, + flush: checkpoints.flush, + reattribute: checkpoints.reattribute, + dispose: (): void => { + blocks.clear() + open.clear() + checkpoints.dispose() + }, + get pending() { + return checkpoints.pending + } + } +} diff --git a/src/main/claude/claude-structured-journal-translation-context-usage.test.ts b/src/main/claude/claude-structured-journal-translation-context-usage.test.ts index 95f2b17f902..534600c71ea 100644 --- a/src/main/claude/claude-structured-journal-translation-context-usage.test.ts +++ b/src/main/claude/claude-structured-journal-translation-context-usage.test.ts @@ -337,7 +337,12 @@ describe('context usage on journal rows', () => { it('counts a turn opening as activity, whatever frame opened it', () => { const onOpen = vi.fn() - const turn = new ClaudeOpenTurn({ sink: journal().sink, settleChildren: () => {}, onOpen }) + const turn = new ClaudeOpenTurn({ + sink: journal().sink, + settleChildren: () => {}, + endOpenWork: () => {}, + onOpen + }) turn.open( { sessionId: 'claude-session', turnId: 'turn-a', startedAt: 1_000, userItemId: 'turn-a' }, 1_000 diff --git a/src/main/claude/claude-structured-journal-translation.test.ts b/src/main/claude/claude-structured-journal-translation.test.ts index 4aa02ed525c..7a9a8d64f56 100644 --- a/src/main/claude/claude-structured-journal-translation.test.ts +++ b/src/main/claude/claude-structured-journal-translation.test.ts @@ -638,7 +638,8 @@ describe('Claude structured journal translation', () => { role: 'reasoning', blocks: [ { type: 'text', text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text } - ] + ], + state: 'completed' }) }) diff --git a/src/main/claude/claude-structured-journal-translation.ts b/src/main/claude/claude-structured-journal-translation.ts index 070798bc521..e2932e0d346 100644 --- a/src/main/claude/claude-structured-journal-translation.ts +++ b/src/main/claude/claude-structured-journal-translation.ts @@ -9,12 +9,14 @@ import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' import { claudeProviderFrameActivity } from '../native-chat/agent-session-wire/provider-frame-activity' import { claudeProviderFrameKind, - createClaudeProviderFrameFallback + createClaudeProviderFrameFallback, + isClaudeProgressFrame } from './claude-structured-provider-fallback' import { taskFrameSentence } from './claude-background-task-frames' import { ClaudeBackgroundTaskRows } from './claude-background-task-rows' import { ClaudeToolOriginRegistry } from './claude-tool-origin-registry' import { ClaudeProvisionalRowCorrections } from './claude-provisional-row-corrections' +import { createClaudeStreamedThinking } from './claude-streamed-thinking' import { ClaudeSubagentRoster } from './claude-subagent-roster' import { ClaudeJournaledRoster } from './claude-subagent-journaled-roster' import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' @@ -76,6 +78,7 @@ export function createClaudeJournalTranslator( const turn = new ClaudeOpenTurn({ sink: deps.sink, settleChildren: (groupKey) => subagents.settleTurn(groupKey), + endOpenWork: (completedAt) => streamedThinking.finishOpen(completedAt), onOpen: () => context.markActivity() }) const context = new ClaudeContextFacts(turn, deps.sink) @@ -135,6 +138,15 @@ export function createClaudeJournalTranslator( deps.sink.publish() } }) + const streamedThinking = createClaudeStreamedThinking({ + ...deps, + producer: subagents.linkage, + turnScope + }) + const flush = (): void => { + streamedText.flush() + streamedThinking.flush() + } const publishActivity = (kind: string, payload: unknown): void => { const turnId = turn.id @@ -148,16 +160,17 @@ export function createClaudeJournalTranslator( } const handleStream = (message: Record, observedAt: number): boolean => { - const delta = streamedBlocks.observe(message) - // `message_start` is the provider's turn boundary. Keep the first text + const delta = streamedBlocks.observe(message, observedAt) + const thinking = streamedThinking.observe(message, observedAt) + // `message_start` is the provider's turn boundary. Keep the first content // delta as a compatibility fallback for streams that omit it. - const source = delta ? claudeStreamTurnSource(message) : claudeStreamTurnStartSource(message) + const source = + delta || thinking ? claudeStreamTurnSource(message) : claudeStreamTurnStartSource(message) turn.ensureOpen(message, source, observedAt) - if (!delta) { - return false + if (delta) { + streamedText.append(delta.identity, delta.text, delta.parentToolUseId) } - streamedText.append(delta.identity, delta.text, delta.parentToolUseId) - return true + return delta !== null || thinking } const messageContext: ClaudeMessageJournalContext = { @@ -165,6 +178,7 @@ export function createClaudeJournalTranslator( tools, streamedBlocks, streamedText, + streamedThinking, subagents, toolOrigins, backgroundTasks, @@ -186,7 +200,7 @@ export function createClaudeJournalTranslator( handle: (event) => { if (event.type === 'ended') { prompts.retryPendingCancellations() - streamedText.flush() + flush() subagents.settleSession() backgroundTasks.settleSession() // The host saw the child end, so the turn's end is observed, not lost. Whether it was a @@ -225,10 +239,14 @@ export function createClaudeJournalTranslator( // writes, so an announcement landing in this same pass has to be visible // to it or the row is stamped provisionally one line too early. const announced = event.type === 'message' && subagents.observeSystemFrame(event.message) - streamedText.flush() + // Only ahead of a frame that can write a row: one per thinking token rewrote the whole row. + if (!(event.type === 'message' && isClaudeProgressFrame(event.message))) { + flush() + } if (announced) { corrections.retry() streamedText.reattribute() + streamedThinking.reattribute() } if (event.type === 'prompt') { prompts.handle(event) @@ -289,12 +307,12 @@ export function createClaudeJournalTranslator( get openTurnInLiveProviderCycle() { return turn.openedInLiveProviderCycle }, - flush: streamedText.flush, + flush, childToolOwner: childQueries.childToolOwner, childActivity: childQueries.childActivity, retryPendingTaskRows: () => backgroundTasks.retryPendingWrites(), get pendingStreamedBlocks() { - return streamedText.pending + return streamedText.pending + streamedThinking.pending }, get contextActivity() { return context.activityRevision @@ -305,9 +323,10 @@ export function createClaudeJournalTranslator( modelMayHaveChanged: () => context.modelMayHaveChanged(), modelWritten: (model) => context.modelWritten(model), dispose: () => { - streamedText.flush() + flush() context.dispose() streamedText.dispose() + streamedThinking.dispose() tools.clear() prompts.clear() streamedBlocks.clear() diff --git a/src/main/claude/claude-structured-launch-resolution.test.ts b/src/main/claude/claude-structured-launch-resolution.test.ts index 9db732d3430..ef75a6ba2b9 100644 --- a/src/main/claude/claude-structured-launch-resolution.test.ts +++ b/src/main/claude/claude-structured-launch-resolution.test.ts @@ -1,6 +1,6 @@ import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' -import { delimiter, join } from 'node:path' +import { delimiter, dirname, join } from 'node:path' import { describe, expect, it, vi } from 'vitest' import type { AgentSessionRecord } from '../../shared/agent-session-record' import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' @@ -13,9 +13,11 @@ import { CLAUDE_SESSION_STATE_EVENTS_ENV, CLAUDE_STRUCTURED_BASE_OPTIONS, claudeSessionIdForOrcaSession, - createClaudeStructuredLaunchResolver + createClaudeStructuredLaunchResolver, + type ClaudeStructuredLaunchResolverDeps } from './claude-structured-launch-resolution' import { claudeStructuredPermissionModeForSettings } from './claude-structured-permission-mode' +import { beginClaudeAuthSwitch, endClaudeAuthSwitch } from '../claude-accounts/live-pty-gate' import { claudeProviderHandle } from '../../shared/agent-session-provider-handle-encoding' const SESSION_ID = 'orca-session-1' @@ -521,3 +523,86 @@ describe('claude structured launch resolution', () => { }) }) }) + +describe('readable Claude thinking', () => { + const launchWith = ( + thinkingDisplay?: ClaudeStructuredLaunchResolverDeps['thinkingDisplay'], + authSwitchSettleTimeoutMs?: number, + command = '/usr/local/bin/claude' + ) => + createClaudeStructuredLaunchResolver({ + // oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: launch resolution reads only getRecord. + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => command, + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ PROJECT_SHIM: '1', ANTHROPIC_API_KEY: 'sk-user' }), + hasTranscript: async () => false, + ...(thinkingDisplay ? { thinkingDisplay } : {}), + ...(authSwitchSettleTimeoutMs === undefined ? {} : { authSwitchSettleTimeoutMs }) + })({ identity: IDENTITY }) + + // Whether the CLI's directory holds a `node` decides if the runtime pairing puts that directory + // first on PATH (Linux CI's /usr/local/bin does, a Mac's usually does not), so both are pinned. + it.each([ + ['without a sibling Node runtime', false], + ['with a sibling Node runtime', true] + ])( + 'probes the CLI the launch runs, on its PATH and shims, without its credentials (%s)', + async (_, sibling) => { + const argsFor = vi.fn( + async (_launch: { command: string; cwd: string; env: Record }) => ({ + 'thinking-display': 'summarized' + }) + ) + const binDir = join(mkdtempSync(join(tmpdir(), 'orca-claude-probe-')), 'bin') + const command = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + makeExecutable(command) + if (sibling) { + makeExecutable(join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node')) + } + const launch = await launchWith({ argsFor }, undefined, command) + const asked = argsFor.mock.calls[0]?.[0] + expect(asked).toMatchObject({ command, cwd: '/repos/workspace-1' }) + const segments = (env: Record | undefined) => + (env?.PATH ?? env?.Path ?? '').split(delimiter) + // The launch's PATH is the probe's plus Orca's own CLI directory, which holds no `claude` or + // runtime, so both resolve the same binary and the same shims in the same order. + const orcaCliDir = launch.env?.ORCA_CLI_COMMAND ? dirname(launch.env.ORCA_CLI_COMMAND) : null + expect(segments(launch.env).filter((dir) => dir !== orcaCliDir)).toEqual(segments(asked?.env)) + expect(segments(asked?.env)[0] === binDir).toBe(sibling) + expect(asked?.env).toMatchObject({ PROJECT_SHIM: '1' }) + expect(asked?.env).not.toHaveProperty('ANTHROPIC_API_KEY') + // The launch keeps the credential the user gave it. + expect(launch.env).toMatchObject({ ANTHROPIC_API_KEY: 'sk-user' }) + expect(launch.options.extraArgs).toEqual({ + 'replay-user-messages': null, + 'thinking-display': 'summarized' + }) + expect(launch.options).not.toHaveProperty('thinking') + } + ) + + it('passes nothing when the CLI is not known to take the flag, or nothing can say', async () => { + const launch = await launchWith({ argsFor: async () => ({}) }) + expect(launch.options.extraArgs).toEqual({ 'replay-user-messages': null }) + expect((await launchWith()).options.extraArgs).toEqual({ 'replay-user-messages': null }) + }) + + it('still rechecks an account switch that began while the probe ran', async () => { + try { + const launch = launchWith( + { + argsFor: async () => { + beginClaudeAuthSwitch() + return {} + } + }, + 10 + ) + await expect(launch).rejects.toMatchObject({ reason: 'accountSwitchInProgress' }) + } finally { + endClaudeAuthSwitch() + } + }) +}) diff --git a/src/main/claude/claude-structured-launch-resolution.ts b/src/main/claude/claude-structured-launch-resolution.ts index e20c1bd8b8e..5f1edeb017a 100644 --- a/src/main/claude/claude-structured-launch-resolution.ts +++ b/src/main/claude/claude-structured-launch-resolution.ts @@ -33,6 +33,7 @@ import { import { resolveClaudeCommand } from '../codex-cli/command' import { resolveSessionFilePath } from '../native-chat/session-file-resolver' import { withoutInheritedClaudeConfigDir } from './claude-config-dir-pin' +import type { ClaudeThinkingDisplaySupport } from './claude-thinking-display-support' import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' import { CLAUDE_STRUCTURED_AGENT } from './claude-structured-agent-definition' @@ -133,6 +134,8 @@ export type ClaudeStructuredLaunchResolverDeps = { authSwitchSettleTimeoutMs?: number /** Account state for the managed-account gate; null when it cannot be read, which refuses. */ readManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null + /** Whether this CLI takes the thinking-display flag. Absent ⇒ the flag is never passed. */ + thinkingDisplay?: Pick /** Whether Claude wrote a transcript for this id; defaults to the transcript resolver. */ hasTranscript?: (input: { providerSessionId: string @@ -152,6 +155,67 @@ async function claudeTranscriptExists(input: { export type ClaudeStructuredInvocation = { command: string; env: Record } +type ClaudeEnvDeps = Pick< + ClaudeStructuredLaunchResolverDeps, + 'resolveCommand' | 'resolveEnv' | 'resolveInheritedEnv' +> + +/** What a child's env is built from, before any auth policy applies to it. */ +export type ClaudeChildEnvSources = { + command: string + overlay: Record | undefined + inheritedEnv: Record +} + +export async function resolveClaudeChildEnvSources( + deps: ClaudeEnvDeps +): Promise { + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const overlay = await deps.resolveEnv?.() + const inheritedEnv = deps.resolveInheritedEnv + ? await deps.resolveInheritedEnv() + : cloneDefinedEnv(process.env) + return { command, overlay: overlay ? cloneDefinedEnv(overlay) : undefined, inheritedEnv } +} + +// Why the overlay merges onto the inherited env rather than replacing it: the child +// still needs PATH and the rest of the shell environment, and withCliRuntimeOnPath +// derives PATH from what it is handed. Ambient Anthropic auth is stripped from the +// inherited half only when a managed account owns the credential; a system-auth +// user's own key is their sign-in and must reach the child. +function claudeChildEnv( + sources: ClaudeChildEnvSources, + stripAuthEnv: boolean, + decorateEnv: (env: Record) => Record = (env) => env +): Record { + return withCliRuntimeOnPath( + sources.command, + decorateEnv({ + ...applyClaudeEnvPatch( + withoutInheritedClaudeConfigDir(sources.inheritedEnv, process.platform), + {}, + { stripAuthEnv, platform: process.platform } + ), + ...sources.overlay + }), + { platform: process.platform } + ) +} + +/** The env a `--version` probe runs with: the launch's own env, PATH and shims included, minus the + * Claude auth variables, auth-like custom headers and an inherited CLAUDE_CONFIG_DIR, which asking + * a version needs none of. Everything else is what this same binary receives at launch anyway. */ +export function claudeProbeEnv(sources: ClaudeChildEnvSources): Record { + return applyClaudeEnvPatch( + claudeChildEnv(sources, true), + {}, + { + stripAuthEnv: true, + platform: process.platform + } + ) +} + /** * The one place a structured Claude child's binary and environment are * resolved. The session launch and the session-less catalog probe both build @@ -160,49 +224,30 @@ export type ClaudeStructuredInvocation = { command: string; env: Record & { authSwitchSettleTimeoutMs?: number }, - decorateEnv: (env: Record) => Record = (env) => env + deps: ClaudeEnvDeps & + Pick & { + authSwitchSettleTimeoutMs?: number + }, + decorateEnv: (env: Record) => Record = (env) => env, + /** Already resolved by a caller that needed them earlier; read again otherwise. */ + resolvedSources?: ClaudeChildEnvSources ): Promise { - const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const sources = resolvedSources ?? (await resolveClaudeChildEnvSources(deps)) const auth = await deps.resolveAuthPolicy() - const overlay = await deps.resolveEnv?.() - const inheritedEnv = deps.resolveInheritedEnv - ? await deps.resolveInheritedEnv() - : cloneDefinedEnv(process.env) // A switch can begin while the policy and overlay resolve, exactly as it can // during the terminal preflight's prepareClaudeAuth — recheck after the awaits. await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) // Under a managed account the pinned credential is the only auth this launch may // use, so an explicit override is refused rather than silently beating the pin. - if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(overlay)) { + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(sources.overlay)) { throw new AgentSessionPreSpawnError(new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE), { reason: 'managedAccountEnvOverride' }) } - // Why the overlay merges onto the inherited env rather than replacing it: the child - // still needs PATH and the rest of the shell environment, and withCliRuntimeOnPath - // derives PATH from what it is handed. Ambient Anthropic auth is stripped from the - // inherited half only when a managed account owns the credential; a system-auth - // user's own key is their sign-in and must reach the child. - const env = withCliRuntimeOnPath( - command, - decorateEnv({ - ...applyClaudeEnvPatch( - withoutInheritedClaudeConfigDir(inheritedEnv, process.platform), - {}, - { - stripAuthEnv: auth.stripAuthEnv, - platform: process.platform - } - ), - ...(overlay ? cloneDefinedEnv(overlay) : {}) - }), - { platform: process.platform } - ) - return { command, env } + return { + command: sources.command, + env: claudeChildEnv(sources, auth.stripAuthEnv, decorateEnv) + } } /** @@ -280,6 +325,14 @@ export function createClaudeStructuredLaunchResolver( ? head.nativeId : claudeSessionIdForOrcaSession(identity.sessionId) const continuesChain = head !== null + const cwd = await deps.resolveWorkspacePath(record.location.workspaceId) + const sources = await resolveClaudeChildEnvSources(deps) + // Asked as soon as the spawn's cwd and PATH are known, so it overlaps what is left to resolve. + const thinkingDisplay = deps.thinkingDisplay?.argsFor({ + command: sources.command, + cwd, + env: claudeProbeEnv(sources) + }) // A start that failed before its first turn wrote no transcript, and `--resume` of an absent // one exits; launch that id fresh instead. With a transcript, `--session-id` would collide. const resumesTranscript = @@ -294,24 +347,33 @@ export function createClaudeStructuredLaunchResolver( const permission = claudeStructuredPermissionOptions( (await deps.resolvePermissionMode?.()) ?? 'default' ) - const { command, env } = await resolveClaudeStructuredInvocation(deps, (base) => - // Every structured session speaks orchestration as itself: its injected id and the Orca CLI. - structuredSessionChildIdentityEnv(record.sessionId, { - ...base, - // The turn translator relies on Claude's authoritative idle frame when no result arrives. - [CLAUDE_SESSION_STATE_EVENTS_ENV]: '1' - }) + const thinkingDisplayArgs = (await thinkingDisplay) ?? {} + // Last: it rechecks the account switch, which may have begun during any await above. + const { command, env } = await resolveClaudeStructuredInvocation( + deps, + (base) => + // Every structured session speaks orchestration as itself: its injected id and the Orca CLI. + structuredSessionChildIdentityEnv(record.sessionId, { + ...base, + // The turn translator relies on Claude's authoritative idle frame when no result arrives. + [CLAUDE_SESSION_STATE_EVENTS_ENV]: '1' + }), + sources ) return { pathToClaudeCodeExecutable: command, options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, ...permission, - extraArgs: { ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs, ...permission.extraArgs }, + extraArgs: { + ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs, + ...permission.extraArgs, + ...thinkingDisplayArgs + }, // Claude owns where a resumed conversation continues; the stored leaf is Orca's bookkeeping. ...(resumesTranscript ? { resume: providerSessionId } : { sessionId: providerSessionId }) }, - cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + cwd, env, claudeConfigDir: record.accountHome.path, providerSessionId, diff --git a/src/main/claude/claude-structured-provider-fallback.ts b/src/main/claude/claude-structured-provider-fallback.ts index af6ed824dd2..7bda9d53799 100644 --- a/src/main/claude/claude-structured-provider-fallback.ts +++ b/src/main/claude/claude-structured-provider-fallback.ts @@ -5,6 +5,7 @@ import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from '../native-chat/agent-session-journal/journal-payload-bounds' import { CLAUDE_STREAM_JSON_FRAME_KINDS } from '../native-chat/agent-session-wire/claude-stream-json-frame-schema' +import { classifyProviderFrame } from '../native-chat/agent-session-wire/provider-frame-disposition' import { type UnhandledProviderFrameJournalItemOptions, readableProviderFrameText, @@ -34,6 +35,24 @@ export function claudeProviderFrameKind(message: Record): strin return ['message', type, subtype ?? eventType].filter(Boolean).join(':') } +// Telemetry the translator never journals: the token tally Claude sends after every thinking delta, +// stream deltas no stream registry carries (signatures, tool input), and keep-alive pings. +const PROGRESS_FRAME_KINDS: ReadonlySet = new Set([ + 'message:system:thinking_tokens', + 'message:stream_event:content_block_delta', + 'message:stream_event:ping' +]) + +/** A frame that writes no row, so nothing streamed has to be journaled ahead of it. A failure it + * reports still surfaces as a row, so it is not one. */ +export function isClaudeProgressFrame(message: Record): boolean { + const kind = claudeProviderFrameKind(message) + return ( + PROGRESS_FRAME_KINDS.has(kind) && + classifyProviderFrame('claude', kind, message) !== 'error-surface' + ) +} + const SETTLED_RESULT_KINDS: ReadonlySet = new Set( CLAUDE_STREAM_JSON_FRAME_KINDS.filter((kind) => kind.startsWith('message:result:')) ) diff --git a/src/main/claude/claude-structured-reasoning-lifecycle.test.ts b/src/main/claude/claude-structured-reasoning-lifecycle.test.ts new file mode 100644 index 00000000000..22376017acd --- /dev/null +++ b/src/main/claude/claude-structured-reasoning-lifecycle.test.ts @@ -0,0 +1,213 @@ +// Every way a streamed Claude reasoning row ends when its final frame never comes. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionAppendOptions } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +type Write = { + key: string + body: AgentJournalItemBody + options: StructuredAgentSessionAppendOptions +} + +function setup() { + const writes: Write[] = [] + const translator = createClaudeJournalTranslator({ + sink: { + appendItem: (identity, body, options) => + writes.push({ key: agentJournalItemKey(identity), body, options }), + appendTombstone: vi.fn(), + publish: vi.fn() + } + }) + const message = ( + frame: Record, + observedAt: number, + startsTurn = false + ): ClaudeStructuredSessionEvent => ({ + type: 'message', + sessionId: 'orca-session', + observedAt, + ...(startsTurn ? { startsTurn: true } : {}), + message: { session_id: 'claude-session', parent_tool_use_id: null, ...frame } + }) + const thinkingDelta = (uuid: string, thinking: string, observedAt: number, index = 0): void => + translator.handle( + message( + { + type: 'stream_event', + uuid, + event: { type: 'content_block_delta', index, delta: { type: 'thinking_delta', thinking } } + }, + observedAt + ) + ) + const reasoningWrites = () => + writes.filter((write) => write.body.kind === 'message' && write.body.role === 'reasoning') + const lastReasoning = () => reasoningWrites().at(-1) + return { writes, translator, message, thinkingDelta, reasoningWrites, lastReasoning } +} + +beforeEach(() => vi.useFakeTimers()) +afterEach(() => vi.useRealTimers()) + +describe('a streamed reasoning row the provider never finishes', () => { + it('ends when the turn settles on its result', () => { + const { translator, message, thinkingDelta, lastReasoning } = setup() + thinkingDelta('delta', 'Unfinished thought', 1_000) + translator.handle( + message({ type: 'result', subtype: 'success', uuid: 'result', is_error: false }, 4_000) + ) + expect(lastReasoning()?.body).toMatchObject({ + blocks: [{ type: 'text', text: 'Unfinished thought' }], + state: 'completed', + completedAt: 4_000 + }) + // The end must survive backpressure: nothing later would write it. + expect(lastReasoning()?.options.lifecycle).toBe(true) + expect(translator.pendingStreamedBlocks).toBe(0) + }) + + it('ends when the CLI reports the session idle', () => { + const { translator, message, thinkingDelta, lastReasoning } = setup() + thinkingDelta('delta', 'Unfinished thought', 1_000) + translator.handle( + message( + { type: 'system', subtype: 'session_state_changed', state: 'idle', uuid: 'idle' }, + 6_000 + ) + ) + expect(lastReasoning()?.body).toMatchObject({ state: 'completed', completedAt: 6_000 }) + }) + + it('ends when the child exits', () => { + const { translator, thinkingDelta, lastReasoning } = setup() + thinkingDelta('delta', 'Unfinished thought', 1_000) + translator.handle({ + type: 'ended', + sessionId: 'orca-session', + reason: 'exit', + observedAt: 7_000 + }) + expect(lastReasoning()?.body).toMatchObject({ state: 'completed', completedAt: 7_000 }) + }) + + it('ends when a new send supersedes its turn', () => { + const { translator, message, thinkingDelta, lastReasoning } = setup() + thinkingDelta('delta', 'Unfinished thought', 1_000) + translator.handle( + message( + { + type: 'user', + uuid: 'user-2', + message: { role: 'user', content: [{ type: 'text', text: 'next' }] } + }, + 8_000, + true + ) + ) + expect(lastReasoning()?.body).toMatchObject({ state: 'completed', completedAt: 8_000 }) + }) + + it('ends when its stream starts a new message instead', () => { + const { translator, message, thinkingDelta, reasoningWrites } = setup() + thinkingDelta('delta', 'Abandoned attempt', 1_000) + translator.flush() + const abandoned = reasoningWrites()[0]?.key + translator.handle( + message( + { + type: 'stream_event', + uuid: 'retry', + event: { type: 'message_start', message: { id: 'm-2' } } + }, + 9_000 + ) + ) + const ends = reasoningWrites().filter((write) => write.key === abandoned) + expect(ends.at(-1)?.body).toMatchObject({ + blocks: [{ type: 'text', text: 'Abandoned attempt' }], + state: 'completed', + completedAt: 9_000 + }) + expect(translator.pendingStreamedBlocks).toBe(0) + }) + + it('writes no end for a block that never had text', () => { + const { translator, message, thinkingDelta, reasoningWrites } = setup() + thinkingDelta('delta', '', 1_000) + translator.handle( + message({ type: 'result', subtype: 'success', uuid: 'result', is_error: false }, 4_000) + ) + expect(reasoningWrites()).toEqual([]) + }) + + it('is not rewritten by the token tally Claude sends after every thinking delta', () => { + const { translator, message, thinkingDelta, reasoningWrites } = setup() + for (let index = 0; index < 20; index += 1) { + thinkingDelta(`delta-${index}`, 'more words ', 1_000 + index) + translator.handle( + message( + { + type: 'system', + subtype: 'thinking_tokens', + estimated_tokens: index, + estimated_tokens_delta: 1, + uuid: `tokens-${index}` + }, + 1_000 + index + ) + ) + } + // Text reaches the row on the coalescer's cadence, not once per tally. + expect(reasoningWrites()).toEqual([]) + vi.advanceTimersByTime(100) + expect(reasoningWrites()).toHaveLength(1) + }) + + // Timings of the first block in a captured 2.1.280 session with summarized display. + it('spans from the block start to its final frame, though text arrives seconds later', () => { + const { translator, message, thinkingDelta, reasoningWrites } = setup() + const stream = (uuid: string, event: Record, observedAt: number): void => + translator.handle(message({ type: 'stream_event', uuid, event }, observedAt)) + stream('start', { type: 'message_start', message: { id: 'm-1' } }, 3_400) + stream( + 'block', + { type: 'content_block_start', index: 0, content_block: { type: 'thinking', thinking: '' } }, + 3_510 + ) + thinkingDelta('delta', 'Planning the module', 8_085) + translator.handle( + message( + { + type: 'assistant', + uuid: 'final', + message: { + id: 'm-1', + role: 'assistant', + content: [{ type: 'thinking', thinking: 'Planning the module', signature: 's' }] + } + }, + 8_160 + ) + ) + const writes = reasoningWrites() + expect(writes.length).toBeGreaterThan(0) + // Every write names the block's start, so whichever creates the row starts it there. + expect(writes.every((write) => write.options.observedAt === 3_510)).toBe(true) + expect(writes.at(-1)?.body).toMatchObject({ state: 'completed', completedAt: 8_160 }) + }) + + it('writes no row for a stream keep-alive, so nothing lands above the open block', () => { + const { translator, message, thinkingDelta, writes } = setup() + thinkingDelta('delta', 'Planning', 1_000) + translator.handle( + message({ type: 'stream_event', uuid: 'ping', event: { type: 'ping' } }, 1_010) + ) + vi.advanceTimersByTime(100) + expect(writes.filter((write) => write.body.kind === 'status')).toEqual([]) + expect(writes.map((write) => write.body.kind)).toEqual(['turn', 'message']) + }) +}) diff --git a/src/main/claude/claude-structured-reasoning.test.ts b/src/main/claude/claude-structured-reasoning.test.ts new file mode 100644 index 00000000000..662b6af1631 --- /dev/null +++ b/src/main/claude/claude-structured-reasoning.test.ts @@ -0,0 +1,218 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function setup() { + const rows = new Map() + const translator = createClaudeJournalTranslator({ + sink: { + // Output also opens its turn; these tests read only the content rows. + appendItem: (identity, body) => { + if (body.kind !== 'turn') { + rows.set(agentJournalItemKey(identity), body) + } + }, + appendTombstone: vi.fn(), + publish: vi.fn() + } + }) + const frame = (message: Record, observedAt?: number): void => + translator.handle({ + type: 'message', + sessionId: 'orca-session', + ...(observedAt === undefined ? {} : { observedAt }), + message: { + session_id: 'session', + parent_tool_use_id: null, + ...message + } + }) + const stream = (uuid: string, event: Record, observedAt?: number): void => + frame({ type: 'stream_event', uuid, event }, observedAt) + const final = (uuid: string, content: unknown[], observedAt?: number): void => + frame( + { + type: 'assistant', + uuid, + message: { id: 'message-1', role: 'assistant', content } + }, + observedAt + ) + return { rows, translator, frame, stream, final } +} + +function reasoning(text: string, lifecycle: Record): AgentJournalItemBody { + return { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text }], + ...lifecycle + } +} + +afterEach(() => vi.useRealTimers()) + +describe('structured Claude reasoning', () => { + it.each(['', ' ', '\n\t'])('omits blank final thinking %j', (thinking) => { + const { rows, final, translator } = setup() + final('final-only', [{ type: 'thinking', thinking }]) + expect([...rows.values()]).toEqual([]) + translator.dispose() + }) + + it('writes final-only thinking closed, with no span it never saw', () => { + const { rows, final, translator } = setup() + final('final-only', [{ type: 'thinking', thinking: 'Inspecting the request' }], 5_000) + expect([...rows.values()]).toEqual([ + reasoning('Inspecting the request', { state: 'completed' }) + ]) + translator.dispose() + }) + + it('streams a running row and closes that same row on its final frame', () => { + vi.useFakeTimers() + const { rows, stream, final, translator } = setup() + stream('start', { type: 'message_start', message: { id: 'message-1' } }, 1_000) + stream( + 'thinking-start', + { type: 'content_block_start', index: 0, content_block: { type: 'thinking', thinking: '' } }, + 1_000 + ) + stream( + 'delta-1', + { + type: 'content_block_delta', + index: 0, + delta: { type: 'thinking_delta', thinking: 'Inspecting ' } + }, + 3_000 + ) + translator.flush() + const firstKey = [...rows.keys()][0] + expect(rows.get(firstKey!)).toEqual(reasoning('Inspecting ', { state: 'running' })) + stream( + 'delta-2', + { + type: 'content_block_delta', + index: 0, + delta: { type: 'thinking_delta', thinking: 'the request' } + }, + 4_000 + ) + translator.flush() + expect([...rows.keys()]).toEqual([firstKey]) + final('thinking-final', [{ type: 'thinking', thinking: 'Inspecting the request' }], 9_000) + expect([...rows.keys()]).toEqual([firstKey]) + expect(rows.get(firstKey!)).toEqual( + reasoning('Inspecting the request', { state: 'completed', completedAt: 9_000 }) + ) + + stream('text-start', { + type: 'content_block_start', + index: 1, + content_block: { type: 'text', text: '' } + }) + stream('text-delta', { + type: 'content_block_delta', + index: 1, + delta: { type: 'text_delta', text: 'Here is the answer' } + }) + translator.flush() + final('text-final', [{ type: 'text', text: 'Here is the answer' }]) + expect([...rows.values()].map((body) => body.kind === 'message' && body.role)).toEqual([ + 'reasoning', + 'assistant' + ]) + expect(translator.pendingStreamedBlocks).toBe(0) + translator.dispose() + }) + + it('closes a streamed row with its streamed text when the final frame carries none', () => { + vi.useFakeTimers() + const { rows, stream, final, translator } = setup() + stream( + 'delta', + { + type: 'content_block_delta', + index: 0, + delta: { type: 'thinking_delta', thinking: 'Summary' } + }, + 1_000 + ) + final('thinking-final', [{ type: 'thinking', thinking: '', signature: 'sig' }], 2_000) + expect([...rows.values()]).toEqual([ + reasoning('Summary', { state: 'completed', completedAt: 2_000 }) + ]) + expect(translator.pendingStreamedBlocks).toBe(0) + translator.dispose() + }) + + it('reconciles an empty final block before the next thinking block', () => { + const { rows, stream, final, translator } = setup() + stream('empty-start', { + type: 'content_block_start', + index: 0, + content_block: { type: 'thinking', thinking: '' } + }) + final('empty-final', [{ type: 'thinking', thinking: '' }]) + stream('next-start', { + type: 'content_block_start', + index: 1, + content_block: { type: 'thinking', thinking: 'Next thought' } + }) + translator.flush() + const key = [...rows.keys()][0] + final('next-final', [{ type: 'thinking', thinking: 'Next thought' }]) + expect([...rows.keys()]).toEqual([key]) + expect(translator.pendingStreamedBlocks).toBe(0) + translator.dispose() + }) + + it('writes streamed thinking into the turn its first delta opened', () => { + vi.useFakeTimers() + const scopes: unknown[] = [] + const translator = createClaudeJournalTranslator({ + sink: { + appendItem: (_identity, body, options) => { + if (body.kind === 'message' && body.role === 'reasoning') { + scopes.push(options.turnScope) + } + }, + appendTombstone: vi.fn(), + publish: vi.fn() + } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { + type: 'stream_event', + session_id: 'session', + parent_tool_use_id: null, + uuid: 'delta', + event: { + type: 'content_block_delta', + index: 0, + delta: { type: 'thinking_delta', thinking: 'Considering' } + } + } + }) + translator.flush() + expect(scopes).toEqual([{ kind: 'turn', turnItemId: expect.stringContaining('turn') }]) + translator.dispose() + }) + + it('omits whitespace-only thinking deltas', () => { + vi.useFakeTimers() + const { rows, stream, translator } = setup() + stream('delta', { + type: 'content_block_delta', + index: 0, + delta: { type: 'thinking_delta', thinking: ' \n ' } + }) + translator.flush() + expect(rows.size).toBe(0) + translator.dispose() + }) +}) diff --git a/src/main/claude/claude-thinking-display-support.test.ts b/src/main/claude/claude-thinking-display-support.test.ts new file mode 100644 index 00000000000..a586a65d70a --- /dev/null +++ b/src/main/claude/claude-thinking-display-support.test.ts @@ -0,0 +1,168 @@ +import { describe, expect, it, vi } from 'vitest' +import { createClaudeThinkingDisplaySupport } from './claude-thinking-display-support' + +const LAUNCH = { command: '/bin/claude', cwd: '/repo', env: { PATH: '/shims' } } +const UNKNOWN_FLAG = new Error( + "claude stream-json exited (code 1): error: unknown option '--thinking-display'" +) + +type Probe = ( + command: string, + launch: { cwd: string; env: Record; timeoutMs: number } +) => Promise + +function supportWith( + probe: Probe, + budgetMs = 20, + keyOf = async (command: string, cwd: string): Promise => `${command}\n${cwd}` +) { + const calls = vi.fn(probe) + // Real time plus whatever a test skips ahead. + let skippedMs = 0 + const support = createClaudeThinkingDisplaySupport({ + probe: calls, + keyOf, + budgetMs, + now: () => performance.now() + skippedMs + }) + return { support, calls, skip: (ms: number) => (skippedMs += ms) } +} + +/** A probe the test answers by hand. */ +function heldProbe() { + let answer: (version: string | null) => void = () => {} + const probe: Probe = () => + new Promise((resolve) => { + answer = resolve + }) + return { probe, answer: (version: string | null) => answer(version) } +} + +describe('the thinking-display flag a launch passes', () => { + it("probes with the launch's own cwd and env and its own kill timeout, then remembers", async () => { + const { support, calls } = supportWith(async () => '2.1.280') + await expect(support.argsFor(LAUNCH)).resolves.toEqual({ 'thinking-display': 'summarized' }) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({ 'thinking-display': 'summarized' }) + expect(calls).toHaveBeenCalledTimes(1) + expect(calls).toHaveBeenCalledWith('/bin/claude', { + cwd: '/repo', + env: { PATH: '/shims' }, + timeoutMs: 10_000 + }) + }) + + it('asks again per workspace: a shim can pick a different CLI there', async () => { + const { support, calls } = supportWith(async () => '2.1.280') + await support.argsFor(LAUNCH) + await support.argsFor({ ...LAUNCH, cwd: '/other' }) + expect(calls).toHaveBeenCalledTimes(2) + }) + + it('passes nothing to a CLI older than the flag, and keeps that answer for good', async () => { + const { support, calls, skip } = supportWith(async () => '2.1.92') + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + skip(60 * 60_000) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + expect(calls).toHaveBeenCalledTimes(1) + }) + + it('asks again after a while when a probe printed no version, failed or was killed', async () => { + for (const probe of [async () => null, () => Promise.reject(new Error('EMFILE'))]) { + const { support, calls, skip } = supportWith(probe) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + // Within the window a hung or broken probe costs no further spawn or wait. + skip(9 * 60_000) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + expect(calls).toHaveBeenCalledTimes(1) + // A failure from a loaded boot heals. + skip(2 * 60_000) + await support.argsFor(LAUNCH) + expect(calls).toHaveBeenCalledTimes(2) + } + }) + + it('waits for a probe that answers within the budget', async () => { + const { support } = supportWith( + () => new Promise((resolve) => setTimeout(() => resolve('2.1.280'), 5)), + 1_000 + ) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({ 'thinking-display': 'summarized' }) + }) + + it('does not wait again on a probe already past its budget, and keeps its late answer', async () => { + const held = heldProbe() + const { support, calls } = supportWith(held.probe, 30) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + const started = performance.now() + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + expect(performance.now() - started).toBeLessThan(25) + held.answer('2.1.280') + await vi.waitFor(async () => + expect(await support.argsFor(LAUNCH)).toEqual({ 'thinking-display': 'summarized' }) + ) + expect(calls).toHaveBeenCalledTimes(1) + }) + + it('stops passing the flag to a binary that exited refusing it', async () => { + const { support, calls } = supportWith(async () => '2.1.280') + await support.argsFor(LAUNCH) + support.observeExit(LAUNCH, UNKNOWN_FLAG) + await vi.waitFor(async () => expect(await support.argsFor(LAUNCH)).toEqual({})) + expect(calls).toHaveBeenCalledTimes(1) + }) + + it('keeps a refusal seen while a probe was still running', async () => { + const held = heldProbe() + const { support } = supportWith(held.probe, 10) + await support.argsFor(LAUNCH) + support.observeExit(LAUNCH, UNKNOWN_FLAG) + await new Promise((resolve) => setTimeout(resolve, 0)) + held.answer('2.1.280') + await new Promise((resolve) => setTimeout(resolve, 0)) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + }) + + it('records nothing for any other startup failure', async () => { + const { support } = supportWith(async () => '2.1.280') + await support.argsFor(LAUNCH) + support.observeExit(LAUNCH, new Error('claude stream-json exited (code 1): not signed in')) + support.observeExit(LAUNCH, new Error("error: unknown option '--thinking'")) + await new Promise((resolve) => setTimeout(resolve, 0)) + await expect(support.argsFor(LAUNCH)).resolves.toEqual({ 'thinking-display': 'summarized' }) + }) + + it('spends the budget finding the binary too, and caches nothing when that runs out', async () => { + let slow = true + const { support, calls } = supportWith( + async () => '2.1.280', + 30, + async (command, cwd) => { + if (slow) { + await new Promise((resolve) => setTimeout(resolve, 1_000)) + } + return `${command}\n${cwd}` + } + ) + const started = performance.now() + await expect(support.argsFor(LAUNCH)).resolves.toEqual({}) + expect(performance.now() - started).toBeLessThan(500) + expect(calls).not.toHaveBeenCalled() + slow = false + await expect(support.argsFor(LAUNCH)).resolves.toEqual({ 'thinking-display': 'summarized' }) + }) + + it('forgets the binaries launched least recently, not the ones in use', async () => { + const { support, calls } = supportWith(async () => '2.1.280') + const at = (index: number) => ({ ...LAUNCH, cwd: `/repo-${index}` }) + for (let index = 1; index <= 32; index += 1) { + await support.argsFor(at(index)) + } + await support.argsFor(at(1)) + await support.argsFor(at(33)) + expect(calls).toHaveBeenCalledTimes(33) + await support.argsFor(at(1)) + expect(calls).toHaveBeenCalledTimes(33) + await support.argsFor(at(2)) + expect(calls).toHaveBeenCalledTimes(34) + }) +}) diff --git a/src/main/claude/claude-thinking-display-support.ts b/src/main/claude/claude-thinking-display-support.ts new file mode 100644 index 00000000000..f7ad53d4433 --- /dev/null +++ b/src/main/claude/claude-thinking-display-support.ts @@ -0,0 +1,175 @@ +import { realpath, stat } from 'node:fs/promises' +import { claudeVersionReaches, probeClaudeCliVersion } from './claude-hook-event-versions' + +// Why: the first CLI whose parser defines the flag (2.1.93 was never published); an older one exits +// on it before the session starts. Found by reading published packages, not by running them. +const CLAUDE_THINKING_DISPLAY_FIRST_VERSION = '2.1.94' + +/** How long a launch waits on a binary nothing is known about yet. A warm probe answers in tens of + * milliseconds; this covers a cold disk, a node install and an antivirus scan, paid once per key + * during a start that already takes seconds. */ +export const CLAUDE_THINKING_DISPLAY_PROBE_BUDGET_MS = 1_500 + +/** A probe still running by now is killed, and its binary gets no flag. */ +const PROBE_KILL_AFTER_MS = 10_000 + +/** How long a probe that gave no version (killed, failed to spawn, unparseable) means "no flag": + * long enough that a hung `--version` costs one wait per stretch, short enough that a failure + * from a loaded boot heals. */ +const FAILED_PROBE_RETRY_AFTER_MS = 10 * 60_000 + +/** Commander's refusal, exactly: any other startup failure says nothing about the flag. */ +const UNKNOWN_FLAG_DIAGNOSTIC = "unknown option '--thinking-display'" + +// One small entry per binary per workspace it launched in; enough for every worktree in active use. +const MAX_REMEMBERED = 32 + +const SUMMARIZED: Readonly> = { 'thinking-display': 'summarized' } + +export type ClaudeThinkingDisplayLaunch = { + command: string + cwd: string + env: Record +} + +type ClaudeVersionProbe = ( + command: string, + launch: { cwd: string; env: Record; timeoutMs: number } +) => Promise + +export type ClaudeThinkingDisplaySupport = { + /** + * Asks for readable thinking: under Orca's launch the CLI otherwise streams thinking blocks with + * no text. Only the display is set, never `--thinking`, so a user who turned thinking off keeps + * it off. A binary not yet known is probed with the launch's own cwd and env, waited on for at + * most the budget from when its probe began; past it the launch goes without the flag. + */ + argsFor: (launch: ClaudeThinkingDisplayLaunch) => Promise>> + /** A child that exited refusing the flag: that binary, in that workspace, never gets it again. */ + observeExit: (launch: Pick, error: Error) => void +} + +/** Which binary a command is in a workspace right now: a shim answers per project, and a + * self-update swaps the link's target or the file. */ +async function claudeBinaryKey(command: string, cwd: string): Promise { + try { + const target = await realpath(command) + return `${target}\n${(await stat(target)).mtimeMs}\n${cwd}` + } catch { + return null + } +} + +/** The promise's value if it settles within `ms`, else undefined. */ +async function within(pending: Promise, ms: number): Promise { + let timer: ReturnType | undefined + const expired = new Promise((resolve) => { + timer = setTimeout(() => resolve(undefined), Math.max(0, ms)) + timer.unref?.() + }) + try { + return await Promise.race([pending, expired]) + } finally { + clearTimeout(timer) + } +} + +export function createClaudeThinkingDisplaySupport( + deps: { + probe: ClaudeVersionProbe + keyOf: (command: string, cwd: string) => Promise + budgetMs: number + now: () => number + } = { + probe: probeClaudeCliVersion, + keyOf: claudeBinaryKey, + budgetMs: CLAUDE_THINKING_DISPLAY_PROBE_BUDGET_MS, + now: () => performance.now() + } +): ClaudeThinkingDisplaySupport { + /** `expiresAt` only on an answer that was no answer: a version or a refusal is final. */ + const known = new Map() + const probing = new Map; startedAt: number }>() + const remember = (key: string, supported: boolean, expiresAt?: number): void => { + known.delete(key) + known.set(key, { supported, ...(expiresAt === undefined ? {} : { expiresAt }) }) + for (const stale of known.keys()) { + if (known.size <= MAX_REMEMBERED) { + break + } + known.delete(stale) + } + } + // A version is kept for the binary's life. A probe that gave none is kept only for a while, so + // a hung CLI costs one wait per stretch and a boot-time failure heals. A refusal seen meanwhile + // wins over either. + const settle = (key: string, supported: boolean, expiresAt?: number): void => { + if (!known.has(key)) { + remember(key, supported, expiresAt) + } + } + const settleWithoutVersion = (key: string): void => + settle(key, false, deps.now() + FAILED_PROBE_RETRY_AFTER_MS) + const lookup = (key: string): boolean | undefined => { + const entry = known.get(key) + if (entry?.expiresAt !== undefined && entry.expiresAt <= deps.now()) { + known.delete(key) + return undefined + } + if (entry) { + // Read as used: the bound drops the binaries launched least recently. + remember(key, entry.supported, entry.expiresAt) + } + return entry?.supported + } + const probe = (key: string, launch: ClaudeThinkingDisplayLaunch) => { + const settled = deps + .probe(launch.command, { cwd: launch.cwd, env: launch.env, timeoutMs: PROBE_KILL_AFTER_MS }) + .then( + (version) => + version === null + ? settleWithoutVersion(key) + : settle(key, claudeVersionReaches(version, CLAUDE_THINKING_DISPLAY_FIRST_VERSION)), + () => settleWithoutVersion(key) + ) + .finally(() => probing.delete(key)) + const started = { settled, startedAt: deps.now() } + probing.set(key, started) + return started + } + + return { + argsFor: async (launch) => { + // The launch never waits longer than the budget, finding the binary included. + const deadline = deps.now() + deps.budgetMs + const key = await within(deps.keyOf(launch.command, launch.cwd), deps.budgetMs) + if (key === undefined || key === null) { + return {} + } + const supported = lookup(key) + if (supported !== undefined) { + return supported ? SUMMARIZED : {} + } + const running = probing.get(key) ?? probe(key, launch) + // The probe's own budget, too: one already past it is not waited on again. + await within( + running.settled, + Math.min(running.startedAt + deps.budgetMs, deadline) - deps.now() + ) + return known.get(key)?.supported === true ? SUMMARIZED : {} + }, + observeExit: (launch, error) => { + if (!error.message.includes(UNKNOWN_FLAG_DIAGNOSTIC)) { + return + } + void deps.keyOf(launch.command, launch.cwd).then((key) => { + if (key !== null) { + remember(key, false) + } + }) + } + } +} + +/** One per process: every structured launch on this host shares what it learned. */ +export const claudeThinkingDisplaySupport = createClaudeThinkingDisplaySupport() diff --git a/src/main/codex/codex-persistent-command-retention.test.ts b/src/main/codex/codex-persistent-command-retention.test.ts index 0f410c6b362..9da3e779207 100644 --- a/src/main/codex/codex-persistent-command-retention.test.ts +++ b/src/main/codex/codex-persistent-command-retention.test.ts @@ -117,6 +117,7 @@ describe('persistent command retention', () => { threadId: `thread-${thread}`, turnId: 'turn', turnLifecycle: null, + completedAt: 1, turnEnd: 'completed', sink, streams: items.streams, @@ -214,6 +215,7 @@ describe('persistent command retention', () => { threadId: 'root', turnId: 'turn', turnLifecycle: null, + completedAt: 1, turnEnd: 'completed', sink, streams: items.streams, diff --git a/src/main/codex/codex-structured-item-stream-bounds.ts b/src/main/codex/codex-structured-item-stream-bounds.ts index 46104c54360..8a7a1f1147f 100644 --- a/src/main/codex/codex-structured-item-stream-bounds.ts +++ b/src/main/codex/codex-structured-item-stream-bounds.ts @@ -1,4 +1,7 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' import { toolExecutionMetadata } from '../../shared/native-chat-tool-identity' +import type { CodexItemStreamState } from './codex-structured-item-stream-contracts' +import type { CodexThreadItem } from './codex-thread-item-identity' export const MAX_CODEX_ITEM_STREAM_STATES = 256 export const MAX_CODEX_ITEM_STREAM_PENDING_PATCHES = 128 @@ -37,3 +40,16 @@ export function boundStreamItem(item: Record): Record boolean handle: ( threadId: string, diff --git a/src/main/codex/codex-structured-item-streams.ts b/src/main/codex/codex-structured-item-streams.ts index 33b3fa7b3cb..291ff544c52 100644 --- a/src/main/codex/codex-structured-item-streams.ts +++ b/src/main/codex/codex-structured-item-streams.ts @@ -5,17 +5,14 @@ import { import { createAgentSessionDeltaCoalescer } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' import { CodexItemStreamRetention } from './codex-item-stream-retention' import { appendCodexItemAndPublish } from './codex-structured-journal-sink' -import { - codexJournalItem, - codexStreamingJournalItem, - type CodexThreadItem -} from './codex-structured-item-translation' +import { codexJournalItem, codexStreamingJournalItem } from './codex-structured-item-translation' +import { withJournalReasoningLifecycle } from '../native-chat/agent-session-journal/journal-reasoning-row' import { codexStructuredItemKey, MAX_CODEX_ITEM_STREAM_PENDING_PATCHES, MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES, MAX_CODEX_ITEM_STREAM_RETAINED_BYTES, - boundStreamItem, + codexItemStreamState, pendingPatchBytes } from './codex-structured-item-stream-bounds' import { @@ -117,8 +114,12 @@ export function createCodexStructuredItemStreams( if (!translated.body) { return true } - return appendCodexItemAndPublish(deps.sink, state.identity, translated.body, attributionOf(key)) - .accepted + // A stream only ever carries an item that has not completed yet. + const body = withJournalReasoningLifecycle(translated.body, { state: 'running' }) + return appendCodexItemAndPublish(deps.sink, state.identity, body, { + ...attributionOf(key), + ...(state.startedAt === undefined ? {} : { observedAt: state.startedAt }) + }).accepted } const persist = (key: string, text: string, force: boolean): boolean => { @@ -209,13 +210,13 @@ export function createCodexStructuredItemStreams( return states.persistentSize }, canTrack: (threadId, item, identity) => - states.canRetain(codexStructuredItemKey(threadId, item.id), { - item: boundStreamItem(item) as CodexThreadItem, - identity - }), - track: (threadId, turnId, item, identity) => { + states.canRetain( + codexStructuredItemKey(threadId, item.id), + codexItemStreamState(item, identity) + ), + track: (threadId, turnId, item, identity, startedAt) => { const key = codexStructuredItemKey(threadId, item.id) - if (!states.retain(key, { item: boundStreamItem(item) as CodexThreadItem, identity })) { + if (!states.retain(key, codexItemStreamState(item, identity, startedAt))) { return false } producers.set(key, { threadId, turnId }) diff --git a/src/main/codex/codex-structured-item-translation.test.ts b/src/main/codex/codex-structured-item-translation.test.ts index ad63d624f58..45a6ca6e9df 100644 --- a/src/main/codex/codex-structured-item-translation.test.ts +++ b/src/main/codex/codex-structured-item-translation.test.ts @@ -851,6 +851,34 @@ describe('codex item bodies', () => { }) }) + it('keeps streamed reasoning as a message and leaves streamed plans as status', () => { + expect(codexStreamingJournalItem({ type: 'reasoning', id: 'r' }, 'thinking')).toEqual({ + handled: true, + body: { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'thinking' }] + } + }) + expect(codexStreamingJournalItem({ type: 'reasoning', id: 'r' }, ' \n ')).toEqual({ + handled: true, + body: null + }) + expect(codexStreamingJournalItem({ type: 'plan', id: 'p' }, 'First\nSecond')).toEqual({ + handled: true, + body: { kind: 'status', text: 'First\nSecond', presentation: 'plan-document' } + }) + }) + + it('omits blank reasoning and preserves the plan document body', () => { + expect(codexItemBody({ type: 'reasoning', id: 'r', text: ' \n ' })).toBeNull() + expect(codexItemBody({ type: 'plan', id: 'p', text: 'First\nSecond' })).toEqual({ + kind: 'status', + text: 'First\nSecond', + presentation: 'plan-document' + }) + }) + it('renders array-shaped reasoning content', () => { expect( codexItemBody({ diff --git a/src/main/codex/codex-structured-item-translation.ts b/src/main/codex/codex-structured-item-translation.ts index 25b821d90d9..50ca92b80ee 100644 --- a/src/main/codex/codex-structured-item-translation.ts +++ b/src/main/codex/codex-structured-item-translation.ts @@ -32,6 +32,7 @@ export { MAX_CODEX_TURN_ORDINAL_BYTES, MAX_CODEX_TURN_ORDINAL_ENTRIES } from './codex-turn-ordinals' +import { journalReasoningBody } from '../native-chat/agent-session-journal/journal-reasoning-row' // Codex thread items → journal item bodies. @@ -76,10 +77,6 @@ export type CodexJournalItem = { handled: boolean } -function reasoningMessageBody(text: string): AgentJournalItemBody { - return { kind: 'message', role: 'reasoning', blocks: [{ type: 'text', text }] } -} - function commandItem(item: CodexThreadItem): CodexJournalItem { const output = readFirstString(item, ['aggregatedOutput', 'aggregated_output']) const bounded = output === null ? null : boundInlineText(output, DEFAULT_JOURNAL_PAYLOAD_LIMITS) @@ -281,13 +278,7 @@ export function codexJournalItem( readTextContent(item, 'text') ?? readTextContent(item, 'summary') ?? readTextContent(item, 'content') - return { - body: - text === null - ? null - : reasoningMessageBody(boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text), - handled: true - } + return { body: journalReasoningBody(text), handled: true } } const unhandled = unhandledProviderFrameJournalItem('codex', `item:${item.type}`, item) return unhandled ? { body: unhandled.body, handled: false } : { body: null, handled: true } @@ -311,6 +302,9 @@ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): if (item.type === 'agentMessage') { return { body: codexStreamingMessageBody(text), handled: true } } + if (item.type === 'reasoning') { + return { body: journalReasoningBody(text), handled: true } + } if (item.type === 'commandExecution') { return commandItem({ ...item, aggregatedOutput: text }) } @@ -334,12 +328,8 @@ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): handled: true } } - const bounded = boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS) return { - body: - item.type === 'reasoning' - ? reasoningMessageBody(bounded.text) - : { kind: 'status', text: bounded.text }, + body: { kind: 'status', text: boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text }, handled: true } } diff --git a/src/main/codex/codex-structured-journal-contracts.ts b/src/main/codex/codex-structured-journal-contracts.ts index af7c1ed3035..b79ac93c3fc 100644 --- a/src/main/codex/codex-structured-journal-contracts.ts +++ b/src/main/codex/codex-structured-journal-contracts.ts @@ -4,8 +4,21 @@ import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-sessio import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' import type { CodexSubagentExecutions } from './codex-subagent-executions' +import type { CodexThreadItem } from './codex-structured-item-translation' +import type { CodexHelperName } from './codex-collab-agent-item-translation' import type { StructuredAgentSessionCommandRun } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +export type CodexActiveJournalItem = { + threadId: string + turnId: string | null + identity: AgentJournalItemIdentity + item: CodexThreadItem + /** Names the helpers a collab call acted on, so a settled revision keeps naming them. */ + helperName?: CodexHelperName + /** Host clock at item/started, for a row whose first write comes later. */ + startedAt?: number +} + export type CodexJournalTranslatorDeps = { sink: StructuredAgentSessionEventSink /** Names this connection in frame-row identities, so a later connection never revises its rows. */ diff --git a/src/main/codex/codex-structured-journal-items.ts b/src/main/codex/codex-structured-journal-items.ts index ec6ac67df5d..17a002225e7 100644 --- a/src/main/codex/codex-structured-journal-items.ts +++ b/src/main/codex/codex-structured-journal-items.ts @@ -1,5 +1,4 @@ import type { - AgentJournalItemBody, AgentJournalItemIdentity, AgentJournalRowAttribution } from '../../shared/agent-session-journal-types' @@ -16,6 +15,7 @@ import type { CodexHelperName } from './codex-collab-agent-item-translation' import { boundStreamItem, codexStructuredItemKey } from './codex-structured-item-stream-bounds' import { codexCommandOutlivesTurn } from './codex-command-lifecycle' import type { + CodexActiveJournalItem, CodexItemTranslation, CodexJournalTranslationAdmission, CodexJournalTranslatorDeps @@ -28,10 +28,19 @@ import { MAX_CODEX_IDENTITY_ENTRIES } from './codex-structured-journal-limits' import { appendCodexLifecycleItem, publishCodexLifecycle } from './codex-structured-journal-sink' -import type { CodexActiveJournalItem } from './codex-structured-journal-settlement' import { readCodexJournalString } from './codex-structured-journal-translation-values' import { readCodexTurnId } from './codex-structured-thread-facts' import { readCodexDispatchEcho } from './codex-structured-dispatch-echo' +import { + codexActiveItemBody, + codexCompletedItem, + interruptedCodexItemBody +} from './codex-unfinished-item-body' +import { + endedJournalReasoning, + withJournalReasoningLifecycle, + type JournalReasoningLifecycle +} from '../native-chat/agent-session-journal/journal-reasoning-row' import type { CodexRowAttribution } from './codex-subagent-linkage' export class CodexJournalItems { @@ -44,7 +53,7 @@ export class CodexJournalItems { constructor( private readonly deps: Pick< CodexJournalTranslatorDeps, - 'sink' | 'coalesceMs' | 'maxRetainedBytes' | 'schedule' + 'sink' | 'coalesceMs' | 'maxRetainedBytes' | 'schedule' | 'now' > & { maxMetadataBytes?: number; attributionFor: CodexRowAttribution }, private readonly activeTurn: (threadId: string) => string | null, private readonly suppress: (threadId: string, turnId: string) => void, @@ -67,7 +76,7 @@ export class CodexJournalItems { } handle( - event: { threadId: string; method: string; params: unknown }, + event: { threadId: string; method: string; params: unknown; observedAt?: number }, source: 'live' | 'history' = 'live' ): CodexItemTranslation { const params = @@ -99,9 +108,23 @@ export class CodexJournalItems { return { handled: true, admission: { accepted: false, reason: 'failed' } } } const itemKey = codexStructuredItemKey(event.threadId, item.id) - const started = - event.method === 'item/completed' ? this.activeItems.get(itemKey)?.item : undefined - const translated = codexJournalItem(item, this.helperName, started) + const active = event.method === 'item/completed' ? this.activeItems.get(itemKey) : undefined + const receivedAt = event.observedAt ?? this.deps.now?.() ?? Date.now() + const lifecycle: JournalReasoningLifecycle = + source === 'history' + ? endedJournalReasoning() + : event.method === 'item/completed' + ? // A completion with no start on record claims no span it never saw. + endedJournalReasoning(active?.startedAt === undefined ? undefined : receivedAt) + : { state: 'running' } + const translated = withItemLifecycle( + codexCompletedItem( + codexJournalItem(item, this.helperName, active?.item), + active, + this.streams + ), + lifecycle + ) const command = readCodexJournalString(item, 'command') if (command) { const boundedCommand = Buffer.from(command, 'utf8') @@ -114,20 +137,22 @@ export class CodexJournalItems { this.streams.forget(event.threadId, item.id) this.activeItems.delete(itemKey) } else { - this.track(event.threadId, turnId, item, identity) - const admission = this.trimActiveState() + const admission = this.trimActiveState(this.growsActiveSet(itemKey, item)) if (!admission.accepted) { return { handled: true, admission } } + // An item whose row waits for its first text still started here. + this.track(event.threadId, turnId, item, identity, receivedAt) } return { handled: true, admission: CODEX_JOURNAL_ADMITTED } } - const admission = this.appendTranslated( - event.method, - identity, - translated, - this.deps.attributionFor(event.threadId, turnId) - ) + // Whichever write creates the row, the row starts with its item: a completion can be the first + // write when its text came within one coalescing window. + const startedAt = event.method === 'item/completed' ? active?.startedAt : receivedAt + const admission = this.appendTranslated(event.method, identity, translated, { + ...this.deps.attributionFor(event.threadId, turnId), + ...(startedAt === undefined ? {} : { observedAt: startedAt }) + }) if (!admission.accepted) { return { handled: true, admission } } @@ -135,11 +160,11 @@ export class CodexJournalItems { this.streams.forget(event.threadId, item.id) this.activeItems.delete(itemKey) } else { - this.track(event.threadId, turnId, item, identity) - const trimAdmission = this.trimActiveState() + const trimAdmission = this.trimActiveState(this.growsActiveSet(itemKey, item)) if (!trimAdmission.accepted) { return { handled: true, admission: trimAdmission } } + this.track(event.threadId, turnId, item, identity, receivedAt) } return { handled: true, admission: CODEX_JOURNAL_ADMITTED } } @@ -155,7 +180,7 @@ export class CodexJournalItems { method: string, identity: AgentJournalItemIdentity, translated: ReturnType, - attribution: AgentJournalRowAttribution + attribution: AgentJournalRowAttribution & { observedAt?: number } ): CodexJournalTranslationAdmission { if (!translated.body) { return CODEX_JOURNAL_ADMITTED @@ -187,17 +212,19 @@ export class CodexJournalItems { threadId: string, turnId: string | null, item: CodexThreadItem, - identity: AgentJournalItemIdentity + identity: AgentJournalItemIdentity, + startedAt?: number ): void { const retainedItem = codexCommandOutlivesTurn(item) ? (boundStreamItem(item) as CodexThreadItem) : item - this.streams.track(threadId, turnId, retainedItem, identity) + this.streams.track(threadId, turnId, retainedItem, identity, startedAt) this.activeItems.set(codexStructuredItemKey(threadId, item.id), { threadId, turnId, identity, item: retainedItem, + ...(startedAt === undefined ? {} : { startedAt }), ...(this.helperName ? { helperName: this.helperName } : {}) }) } @@ -229,8 +256,15 @@ export class CodexJournalItems { return identity } - private trimActiveState(): CodexJournalTranslationAdmission { - while (this.activeItems.size - this.streams.persistentCount > MAX_CODEX_ACTIVE_ITEMS) { + private growsActiveSet(itemKey: string, item: CodexThreadItem): boolean { + return !this.activeItems.has(itemKey) && !codexCommandOutlivesTurn(item) + } + + /** Makes room for an item before it is tracked: tracking it first lets the stream bound drop an + * evictee's text before this eviction can close the evictee's row with it. */ + private trimActiveState(incoming: boolean): CodexJournalTranslationAdmission { + const room = MAX_CODEX_ACTIVE_ITEMS - (incoming ? 1 : 0) + while (this.activeItems.size - this.streams.persistentCount > room) { const oldest = [...this.activeItems].find( ([, active]) => !codexCommandOutlivesTurn(active.item) )?.[0] @@ -239,14 +273,12 @@ export class CodexJournalItems { } const evicted = this.activeItems.get(oldest) if (evicted) { - const translated = codexJournalItem(evicted.item, this.helperName).body - if (translated) { - const admission = appendCodexLifecycleItem( - this.deps.sink, - evicted.identity, - evictedActiveBody(translated), - this.deps.attributionFor(evicted.threadId, evicted.turnId) - ) + const body = interruptedCodexItemBody(codexActiveItemBody(evicted, this.streams)) + if (body) { + const admission = appendCodexLifecycleItem(this.deps.sink, evicted.identity, body, { + ...this.deps.attributionFor(evicted.threadId, evicted.turnId), + ...(evicted.startedAt === undefined ? {} : { observedAt: evicted.startedAt }) + }) if (!admission.accepted) { return admission } @@ -264,23 +296,11 @@ export class CodexJournalItems { } } -function evictedActiveBody(body: AgentJournalItemBody): AgentJournalItemBody { - if (body.kind === 'tool-call' && body.state === 'running') { - return { ...body, state: 'failed' } - } - if ( - (body.kind === 'approval' || body.kind === 'question') && - body.resolution.state === 'pending' - ) { - return { - ...body, - resolution: { - state: 'cancelled', - selectedOptionId: null, - resolvedBy: null, - resolvedAt: null - } - } - } - return body +function withItemLifecycle( + translated: ReturnType, + lifecycle: JournalReasoningLifecycle +): ReturnType { + return translated.body + ? { ...translated, body: withJournalReasoningLifecycle(translated.body, lifecycle) } + : translated } diff --git a/src/main/codex/codex-structured-journal-settlement.ts b/src/main/codex/codex-structured-journal-settlement.ts index e24cbd87037..857e82dc33f 100644 --- a/src/main/codex/codex-structured-journal-settlement.ts +++ b/src/main/codex/codex-structured-journal-settlement.ts @@ -1,4 +1,3 @@ -import { endedRunningAgentJournalToolCall } from '../../shared/agent-journal-tool-call-lifecycle' import { AGENT_JOURNAL_THREAD_SCOPE, type AgentJournalItemBody, @@ -15,14 +14,8 @@ import type { StructuredAgentSessionSinkAdmission } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' import { cancelledJournalPromptBody } from '../native-chat/agent-session-journal/journal-prompt-body-bounds' -import { - codexJournalItem, - codexStreamingJournalItem, - type CodexThreadItem, - type CodexTurnOrdinals -} from './codex-structured-item-translation' +import type { CodexTurnOrdinals } from './codex-structured-item-translation' import type { CodexStructuredItemStreams } from './codex-structured-item-streams' -import type { CodexHelperName } from './codex-collab-agent-item-translation' import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' import { codexCommandOutlivesTurn } from './codex-command-lifecycle' import { @@ -30,16 +23,9 @@ import { codexTurnLifecycleIdentity } from './codex-structured-journal-translation-turns' import { appendCodexLifecycleMutations } from './codex-structured-journal-sink' +import { codexActiveItemBody, interruptedCodexItemBody } from './codex-unfinished-item-body' import type { CodexRowAttribution } from './codex-subagent-linkage' - -export type CodexActiveJournalItem = { - threadId: string - turnId: string | null - identity: AgentJournalItemIdentity - item: CodexThreadItem - /** Names the helpers a collab call acted on, so a settled revision keeps naming them. */ - helperName?: CodexHelperName -} +import type { CodexActiveJournalItem } from './codex-structured-journal-contracts' export type CodexPendingJournalPrompt = { threadId: string @@ -63,17 +49,17 @@ export function settleCodexJournalSession(input: { * conversation command claimed, whose record the host settles. */ settledTurnLifecycle: (threadId: string, turnId: string) => AgentJournalTurnLifecycle | null attributionFor: CodexRowAttribution + now?: () => number }): StructuredAgentSessionSinkAdmission { // Rows from every thread settle in this one batch, so each names its own producer. const mutations: JournalLifecycleMutationInput[] = [] const turnOrdinalsToForget: { threadId: string; turnId: string }[] = [] for (const active of input.activeItems.values()) { - const streamed = input.streams.snapshot(active.threadId, active.item.id) - const translated = streamed - ? codexStreamingJournalItem(active.item, streamed.text) - : codexJournalItem(active.item, active.helperName) // The host saw the child go, so its work was cut short. - const body = settledActiveBody(translated.body, 'interrupted') + const body = interruptedCodexItemBody(codexActiveItemBody(active, input.streams), { + at: input.event.observedAt ?? input.now?.() ?? Date.now(), + call: 'interrupted' + }) if (body) { mutations.push(settledRow(input.attributionFor, active, body)) } @@ -121,6 +107,8 @@ export function settleCodexJournalTurn(input: { turnId: string /** Null off the primary thread: only the primary turn owns a lifecycle row. */ turnLifecycle: AgentJournalTurnLifecycle | null + /** Host clock when the turn's end arrived, which is also the end of anything it left open. */ + completedAt: number /** How Codex ended the turn, on every thread: what a call it left running became. */ turnEnd: Extract sink: StructuredAgentSessionEventSink @@ -143,11 +131,10 @@ export function settleCodexJournalTurn(input: { if (codexCommandOutlivesTurn(active.item)) { continue } - const streamed = input.streams.snapshot(active.threadId, active.item.id) - const translated = streamed - ? codexStreamingJournalItem(active.item, streamed.text) - : codexJournalItem(active.item, active.helperName) - const body = settledActiveBody(translated.body, input.turnEnd) + const body = interruptedCodexItemBody(codexActiveItemBody(active, input.streams), { + at: input.completedAt, + call: input.turnEnd + }) if (body) { mutations.push(settledRow(input.attributionFor, active, body)) } @@ -201,24 +188,6 @@ function settledRow( return journalLifecycleItemMutation(attributionFor(row.threadId, row.turnId), row.identity, body) } -function settledActiveBody( - body: AgentJournalItemBody | null, - turnEnd: 'completed' | 'interrupted' -): AgentJournalItemBody | null { - if (!body) { - return null - } - if (body.kind === 'tool-call') { - return endedRunningAgentJournalToolCall(body, turnEnd) - } - if (body.kind === 'message') { - return body - } - return body.kind === 'diff' - ? { kind: 'status', text: 'File changes were interrupted before completion.' } - : body -} - function exitSettlementId(event: Extract): string { const fence = 'fence' in event ? event.fence : 0 const generation = 'acquisitionGeneration' in event ? event.acquisitionGeneration : 'legacy' diff --git a/src/main/codex/codex-structured-journal-sink.ts b/src/main/codex/codex-structured-journal-sink.ts index 0823e5679cf..dc4778ea350 100644 --- a/src/main/codex/codex-structured-journal-sink.ts +++ b/src/main/codex/codex-structured-journal-sink.ts @@ -99,7 +99,8 @@ export function appendCodexLifecycleItem( sink: StructuredAgentSessionEventSink, identity: AgentJournalItemIdentity, body: AgentJournalItemBody, - attribution: AgentJournalRowAttribution + /** With the host time a row first written here should carry instead of its append time. */ + attribution: AgentJournalRowAttribution & { observedAt?: number } ): CodexJournalTranslationAdmission { const options = { lifecycle: true, ...attribution } if (sink.tryAppendItem) { diff --git a/src/main/codex/codex-structured-journal-translation-settlement.test.ts b/src/main/codex/codex-structured-journal-translation-settlement.test.ts index 6c08f01bc49..a03ee84e307 100644 --- a/src/main/codex/codex-structured-journal-translation-settlement.test.ts +++ b/src/main/codex/codex-structured-journal-translation-settlement.test.ts @@ -834,7 +834,8 @@ describe('codex journal translation', () => { expect(reduced.get('orca:codex-item%3Athread-abc%3Ar-1')).toEqual({ kind: 'message', role: 'reasoning', - blocks: [{ type: 'text', text: 'thinking' }] + blocks: [{ type: 'text', text: 'thinking' }], + state: 'running' }) expect(reduced.get('orca:codex-item%3Athread-abc%3Apatch-1')).toMatchObject({ kind: 'diff', diff --git a/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts b/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts index 5c930eb9f92..d9c52fa181a 100644 --- a/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts +++ b/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts @@ -193,6 +193,7 @@ export class CodexJournalTurnBoundaries { threadId: event.threadId, turnId, turnLifecycle, + completedAt, turnEnd: codexTurnLifecycleState(status), streams: this.deps.items.streams, activeItems: this.deps.items.activeItems, diff --git a/src/main/codex/codex-structured-journal-translation.ts b/src/main/codex/codex-structured-journal-translation.ts index 20fd8bf58df..19fee37a536 100644 --- a/src/main/codex/codex-structured-journal-translation.ts +++ b/src/main/codex/codex-structured-journal-translation.ts @@ -167,7 +167,8 @@ export function createCodexJournalTranslator( completedAt: event.observedAt ?? deps.now?.() ?? Date.now() }) : null, - attributionFor + attributionFor, + ...(deps.now ? { now: deps.now } : {}) }) if (!admission.accepted) { return admission diff --git a/src/main/codex/codex-structured-reasoning-lifecycle.test.ts b/src/main/codex/codex-structured-reasoning-lifecycle.test.ts new file mode 100644 index 00000000000..38fbc2f876b --- /dev/null +++ b/src/main/codex/codex-structured-reasoning-lifecycle.test.ts @@ -0,0 +1,267 @@ +// Every way a Codex reasoning row opens and ends. +import { describe, expect, it } from 'vitest' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { + AGENT_JOURNAL_THREAD_SCOPE, + type AgentJournalItemBody +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { CodexJournalItems } from './codex-structured-journal-items' +import { MAX_CODEX_ACTIVE_ITEMS } from './codex-structured-journal-limits' +import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' +import { createCodexJournalTranslator } from './codex-structured-journal-translation' + +const SESSION_ID = 'session-1' +const THREAD_ID = 'thread-abc' +const TURN_ID = 'turn-1' +const ROW = 'orca:codex-item%3Athread-abc%3Ar-1' + +function recorder() { + const rows = new Map() + /** The host time each row's first write asked to be stamped with. */ + const firstObservedAt = new Map() + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body, options) => { + const key = agentJournalItemKey(identity) + if (!rows.has(key)) { + firstObservedAt.set(key, options.observedAt) + } + rows.set(key, body) + }, + appendTombstone: () => {}, + publish: () => {} + } + return { rows, sink, firstObservedAt } +} + +function notification( + method: string, + params: unknown, + observedAt?: number +): CodexStructuredSessionEvent { + return { + type: 'notification', + sessionId: SESSION_ID, + threadId: THREAD_ID, + method, + params, + ...(observedAt === undefined ? {} : { observedAt }) + } +} + +function reasoning(id: string, summary: string[] = []) { + return { item: { type: 'reasoning', id, summary, content: [] } } +} + +function streamingTurn() { + const { rows, sink, firstObservedAt } = recorder() + const translator = createCodexJournalTranslator({ + sink, + primaryThreadId: () => THREAD_ID, + sessionId: SESSION_ID, + // Deltas are written as they arrive, so the open row is visible without a timer. + schedule: (run) => { + run() + return () => {} + } + }) + translator.handle(notification('turn/started', { turn: { id: TURN_ID } }, 1_000)) + translator.handle(notification('item/started', reasoning('r-1'), 2_000)) + // Production times only turn and item boundaries at receipt; a delta carries no time. + translator.handle( + notification('item/reasoning/summaryTextDelta', { itemId: 'r-1', delta: 'Planning' }) + ) + return { rows, translator, firstObservedAt } +} + +/** A turn whose deltas wait out the coalescing window, as they do in production. */ +function coalescedTurn(now?: () => number) { + const { rows, sink, firstObservedAt } = recorder() + const translator = createCodexJournalTranslator({ + sink, + primaryThreadId: () => THREAD_ID, + sessionId: SESSION_ID, + schedule: () => () => {}, + ...(now ? { now } : {}) + }) + translator.handle(notification('turn/started', { turn: { id: TURN_ID } }, 1_000)) + return { rows, translator, firstObservedAt } +} + +describe('a Codex reasoning row', () => { + it('writes no row while its item has no text, and opens with its first summary text', () => { + const { rows, sink } = recorder() + const translator = createCodexJournalTranslator({ sink, primaryThreadId: () => THREAD_ID }) + translator.handle(notification('turn/started', { turn: { id: TURN_ID } }, 1_000)) + translator.handle(notification('item/started', reasoning('r-1'), 2_000)) + expect(rows.get(ROW)).toBeUndefined() + const streamed = streamingTurn() + expect(streamed.rows.get(ROW)).toMatchObject({ state: 'running' }) + }) + + it('starts when its item started, though its first text came later', () => { + const { firstObservedAt } = streamingTurn() + expect(firstObservedAt.get(ROW)).toBe(2_000) + }) + + // Captured: summary text lands 13–50 ms before item/completed, inside one coalescing window. + it('starts at item/started when its first write is the completion itself', () => { + const streamed = coalescedTurn() + streamed.translator.handle(notification('item/started', reasoning('r-1'), 2_000)) + streamed.translator.handle( + notification('item/reasoning/summaryTextDelta', { itemId: 'r-1', delta: 'Planning' }) + ) + streamed.translator.handle( + notification('item/completed', reasoning('r-1', ['Planning']), 7_650) + ) + expect(streamed.firstObservedAt.get(ROW)).toBe(2_000) + expect(streamed.rows.get(ROW)).toMatchObject({ state: 'completed', completedAt: 7_650 }) + + const completedOnly = coalescedTurn() + completedOnly.translator.handle(notification('item/started', reasoning('r-1'), 2_000)) + completedOnly.translator.handle( + notification('item/completed', reasoning('r-1', ['Planning']), 7_650) + ) + expect(completedOnly.firstObservedAt.get(ROW)).toBe(2_000) + }) + + it('times its start and end on the host clock when a boundary arrives without a time', () => { + let now = 2_000 + const { translator, rows, firstObservedAt } = coalescedTurn(() => now) + translator.handle(notification('item/started', reasoning('r-1'))) + now = 6_000 + translator.handle(notification('item/completed', reasoning('r-1', ['Planning']))) + expect(firstObservedAt.get(ROW)).toBe(2_000) + expect(rows.get(ROW)).toMatchObject({ completedAt: 6_000 }) + }) + + it('ends when its item completes, at the completion it saw', () => { + const { rows, translator } = streamingTurn() + translator.handle(notification('item/completed', reasoning('r-1', ['Planning']), 5_000)) + expect(rows.get(ROW)).toEqual({ + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Planning' }], + state: 'completed', + completedAt: 5_000 + }) + }) + + it('ends with its turn when the item never completes', () => { + const { rows, translator } = streamingTurn() + translator.handle(notification('turn/completed', { turn: { id: TURN_ID } }, 6_000)) + expect(rows.get(ROW)).toMatchObject({ state: 'completed', completedAt: 6_000 }) + }) + + it('ends when the provider exits mid-item', () => { + const { rows, translator } = streamingTurn() + translator.handle({ type: 'ended', sessionId: SESSION_ID, reason: 'exit', observedAt: 7_000 }) + expect(rows.get(ROW)).toMatchObject({ state: 'completed', completedAt: 7_000 }) + }) + + it('replays from history as ended, with no span it never saw', () => { + const { rows, sink } = recorder() + const items = new CodexJournalItems( + { sink, attributionFor: () => ({ turnScope: AGENT_JOURNAL_THREAD_SCOPE }) }, + () => TURN_ID, + () => {} + ) + items.handle( + { threadId: THREAD_ID, method: 'item/completed', params: reasoning('r-1', ['Planned']) }, + 'history' + ) + expect(rows.get(ROW)).toMatchObject({ state: 'completed' }) + expect(rows.get(ROW)).not.toHaveProperty('completedAt') + }) + + it('ends with its streamed text and no claimed time when the bounded live set drops it', () => { + const { rows, sink } = recorder() + const items = new CodexJournalItems( + { + sink, + attributionFor: () => ({ turnScope: AGENT_JOURNAL_THREAD_SCOPE }), + schedule: () => () => {} + }, + () => TURN_ID, + () => {} + ) + // As captured: a started reasoning item carries an empty summary; its text only streams. + items.handle({ threadId: THREAD_ID, method: 'item/started', params: reasoning('r-1') }) + items.streams.handle(THREAD_ID, 'item/reasoning/summaryTextDelta', { + itemId: 'r-1', + delta: 'Thinking' + }) + expect(rows.get(ROW)).toBeUndefined() + for (let index = 2; index <= MAX_CODEX_ACTIVE_ITEMS + 1; index += 1) { + items.handle({ threadId: THREAD_ID, method: 'item/started', params: reasoning(`r-${index}`) }) + } + expect(rows.get(ROW)).toEqual({ + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Thinking' }], + state: 'completed' + }) + }) + + it('ends with its streamed text when the completion itself carries none', () => { + const { translator, rows } = coalescedTurn() + translator.handle(notification('item/started', reasoning('r-1'), 2_000)) + translator.handle( + notification('item/reasoning/summaryTextDelta', { itemId: 'r-1', delta: '**Plan**' }) + ) + translator.handle(notification('item/completed', reasoning('r-1'), 7_000)) + expect(rows.get(ROW)).toEqual({ + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: '**Plan**' }], + state: 'completed', + completedAt: 7_000 + }) + }) + + it('claims no span for a completion whose start was never seen', () => { + const { translator, rows } = coalescedTurn() + translator.handle(notification('item/completed', reasoning('r-1', ['Planned']), 7_000)) + expect(rows.get(ROW)).toMatchObject({ state: 'completed' }) + expect(rows.get(ROW)).not.toHaveProperty('completedAt') + }) + + it('leaves an evicted file change what a settle would, not its command output as a patch', () => { + const { rows, sink } = recorder() + const items = new CodexJournalItems( + { + sink, + attributionFor: () => ({ turnScope: AGENT_JOURNAL_THREAD_SCOPE }), + schedule: () => () => {} + }, + () => TURN_ID, + () => {} + ) + const changes = [{ path: 'src/app.ts', kind: { type: 'update' }, diff: '@@ -1 +1 @@' }] + items.handle({ + threadId: THREAD_ID, + method: 'item/started', + params: { item: { type: 'fileChange', id: 'patch-1', changes, status: 'inProgress' } } + }) + items.streams.handle(THREAD_ID, 'item/fileChange/outputDelta', { + itemId: 'patch-1', + delta: 'Success. Updated the following files:' + }) + for (let index = 2; index <= MAX_CODEX_ACTIVE_ITEMS + 1; index += 1) { + items.handle({ threadId: THREAD_ID, method: 'item/started', params: reasoning(`r-${index}`) }) + } + expect(rows.get('orca:codex-item%3Athread-abc%3Apatch-1')).toEqual({ + kind: 'status', + text: 'File changes were interrupted before completion.' + }) + }) + + it('keeps the start of an item whose started frame already carried text', () => { + const { translator, rows, firstObservedAt } = coalescedTurn() + translator.handle(notification('item/started', reasoning('r-1', ['Plan']), 2_000)) + expect(rows.get(ROW)).toMatchObject({ state: 'running' }) + translator.handle(notification('item/completed', reasoning('r-1', ['Plan']), 7_000)) + expect(firstObservedAt.get(ROW)).toBe(2_000) + expect(rows.get(ROW)).toMatchObject({ state: 'completed', completedAt: 7_000 }) + }) +}) diff --git a/src/main/codex/codex-structured-session-acquire.ts b/src/main/codex/codex-structured-session-acquire.ts index 170c142ba43..61856ef5f24 100644 --- a/src/main/codex/codex-structured-session-acquire.ts +++ b/src/main/codex/codex-structured-session-acquire.ts @@ -38,6 +38,7 @@ import { } from './codex-structured-fast-mode' import { assertCodexConnectionOpen, + CODEX_RECEIPT_TIMED_METHODS, codexSessionLifecycle, mintCodexAcquisitionGeneration, type CodexAcquisitionRegistry, @@ -49,8 +50,6 @@ import type { CodexStructuredSessionTeardown } from './codex-structured-session- import type { CodexStructuredNotificationRetry } from './codex-structured-notification-retry' import type { deliverCodexServerRequest } from './codex-structured-provider-events' -const TURN_BOUNDARIES: ReadonlySet = new Set(['turn/started', 'turn/completed']) - export async function acquireCodexStructuredSession(input: { input: StructuredAgentSessionAcquireInput deps: CodexStructuredSessionAdapterDeps @@ -144,7 +143,9 @@ export async function acquireCodexStructuredSession(input: { { onNotification: (method, params) => { // Stamped at receipt, ahead of any pre-publication buffering or retry. - const observedAt = TURN_BOUNDARIES.has(method) ? (deps.now?.() ?? Date.now()) : undefined + const observedAt = CODEX_RECEIPT_TIMED_METHODS.has(method) + ? (deps.now?.() ?? Date.now()) + : undefined const dispatchSequenceAtReceipt = method === 'turn/started' ? dispatchEchoes.latestSequence() : undefined input.deliver( diff --git a/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts b/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts index 3066b561336..97821f36c92 100644 --- a/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts +++ b/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts @@ -207,6 +207,29 @@ describe('CodexStructuredSessionAdapter lifecycle', () => { expect(adapter.backgroundTaskStops('session-1')).toBeDefined() }) + it('times turn and item boundaries at receipt, and nothing else', async () => { + const codex = fakeCodex() + const events: CodexStructuredSessionEvent[] = [] + await acquired(codex, {}, events) + const connection = codex.connections[0] + const item = { type: 'reasoning', id: 'r-1', summary: [], content: [] } + connection.handlers.onNotification?.('item/started', { threadId: THREAD_ID, item }) + connection.handlers.onNotification?.('item/reasoning/summaryTextDelta', { + threadId: THREAD_ID, + itemId: 'r-1', + delta: 'Planning' + }) + connection.handlers.onNotification?.('item/completed', { threadId: THREAD_ID, item }) + const timed = events.flatMap((event) => + event.type === 'notification' ? [[event.method, event.observedAt]] : [] + ) + expect(timed).toEqual([ + ['item/started', 1_700_000_000_500], + ['item/reasoning/summaryTextDelta', undefined], + ['item/completed', 1_700_000_000_500] + ]) + }) + it('ignores Codex traffic that arrives after the session is gone', async () => { const codex = fakeCodex() const adapter = await acquired(codex) diff --git a/src/main/codex/codex-structured-session-state.ts b/src/main/codex/codex-structured-session-state.ts index 5773b659f5e..7669895937d 100644 --- a/src/main/codex/codex-structured-session-state.ts +++ b/src/main/codex/codex-structured-session-state.ts @@ -45,6 +45,15 @@ export type CodexStructuredLaunch = { env?: Record } +/** Turn and item boundaries, timed by when the host received them, never by when a buffered or + * retried delivery got round to them. */ +export const CODEX_RECEIPT_TIMED_METHODS: ReadonlySet = new Set([ + 'turn/started', + 'turn/completed', + 'item/started', + 'item/completed' +]) + export type CodexStructuredSessionEvent = | { type: 'notification' @@ -52,7 +61,8 @@ export type CodexStructuredSessionEvent = threadId: string method: string params: unknown - /** Host receipt time of a turn boundary; survives retry and deferral so a replay is not re-stamped. */ + /** Host receipt time of a `CODEX_RECEIPT_TIMED_METHODS` boundary; survives retry and + * deferral so a replay is not re-stamped. */ observedAt?: number /** Highest dispatch sequence armed when this turn-start was first received. */ dispatchSequenceAtReceipt?: number diff --git a/src/main/codex/codex-unfinished-item-body.ts b/src/main/codex/codex-unfinished-item-body.ts new file mode 100644 index 00000000000..0234f194925 --- /dev/null +++ b/src/main/codex/codex-unfinished-item-body.ts @@ -0,0 +1,72 @@ +import { + endedRunningAgentJournalToolCall, + type AgentJournalRunningCallEnd +} from '../../shared/agent-journal-tool-call-lifecycle' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import { cancelledJournalPromptBody } from '../native-chat/agent-session-journal/journal-prompt-body-bounds' +import { + endedJournalReasoning, + withJournalReasoningLifecycle +} from '../native-chat/agent-session-journal/journal-reasoning-row' +import type { CodexStructuredItemStreams } from './codex-structured-item-stream-contracts' +import { + codexJournalItem, + codexStreamingJournalItem, + type CodexJournalItem +} from './codex-structured-item-translation' +import type { CodexActiveJournalItem } from './codex-structured-journal-contracts' + +/** What an item still open is built from, whoever ends it — its turn, its provider, the bounded + * live set, or a completion that carried nothing: the text streamed so far when there is any, + * else the item as it started, which usually carries none. */ +export function codexActiveItemBody( + active: CodexActiveJournalItem, + streams: Pick +): AgentJournalItemBody | null { + const streamed = streams.snapshot(active.threadId, active.item.id) + return ( + streamed + ? codexStreamingJournalItem(active.item, streamed.text) + : codexJournalItem(active.item, active.helperName) + ).body +} + +/** A reasoning completion that carried no text of its own still ends the row its stream wrote, + * from the same text a settle would use. */ +export function codexCompletedItem( + completed: CodexJournalItem, + active: CodexActiveJournalItem | undefined, + streams: Pick +): CodexJournalItem { + if (completed.body || !active || active.item.type !== 'reasoning') { + return completed + } + return { body: codexActiveItemBody(active, streams), handled: true } +} + +/** The row an item that will never complete is left with: a running call ended as its turn or + * session did (`end.call`; failed when nothing says), a patch said to be interrupted, a prompt + * cancelled, a message ended — at `end.at` when the host saw the end. */ +export function interruptedCodexItemBody( + body: AgentJournalItemBody | null, + end: { at?: number; call?: AgentJournalRunningCallEnd } = {} +): AgentJournalItemBody | null { + if (!body) { + return null + } + if (body.kind === 'tool-call') { + return end.call + ? endedRunningAgentJournalToolCall(body, end.call) + : { ...body, state: 'failed' } + } + if (body.kind === 'message') { + return withJournalReasoningLifecycle(body, endedJournalReasoning(end.at)) + } + if (body.kind === 'diff') { + return { kind: 'status', text: 'File changes were interrupted before completion.' } + } + return (body.kind === 'approval' || body.kind === 'question') && + body.resolution.state === 'pending' + ? (cancelledJournalPromptBody(body) ?? body) + : body +} diff --git a/src/main/native-chat/agent-session-journal/journal-reasoning-row.ts b/src/main/native-chat/agent-session-journal/journal-reasoning-row.ts new file mode 100644 index 00000000000..e8aa2e73664 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-reasoning-row.ts @@ -0,0 +1,39 @@ +import type { + AgentJournalItemBody, + AgentJournalMessageItem +} from '../../../shared/agent-session-journal-types' +import { boundInlineText, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' + +export type JournalReasoningLifecycle = Pick + +/** Null for blank reasoning: a block or item with no readable text journals no row. */ +export function journalReasoningBody( + text: string | null, + lifecycle: JournalReasoningLifecycle = {} +): AgentJournalMessageItem | null { + return text?.trim() + ? { + kind: 'message', + role: 'reasoning', + blocks: [ + { type: 'text', text: boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text } + ], + ...lifecycle + } + : null +} + +/** Stamps a reasoning row with whether its block or item is still open, as the caller knows it; + * every other body passes through untouched. */ +export function withJournalReasoningLifecycle( + body: AgentJournalItemBody, + lifecycle: JournalReasoningLifecycle +): AgentJournalItemBody { + return body.kind === 'message' && body.role === 'reasoning' ? { ...body, ...lifecycle } : body +} + +/** A reasoning row's end: `completedAt` is when the host saw it end, or the turn or exit that cut + * it off; absent when no end was seen live. */ +export function endedJournalReasoning(completedAt?: number): JournalReasoningLifecycle { + return { state: 'completed', ...(completedAt === undefined ? {} : { completedAt }) } +} diff --git a/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts index 073dcd12bc7..0e66d72bc13 100644 --- a/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts +++ b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts @@ -4,6 +4,7 @@ import { } from '../../../shared/agent-journal-tool-call-lifecycle' import type { AgentJournalItemBody, + AgentJournalMessageItem, AgentJournalTurnScope } from '../../../shared/agent-session-journal-types' import { @@ -11,6 +12,7 @@ import { readAgentJournalTurn } from '../../../shared/agent-session-turn-record' import { cancelledJournalPromptBody } from './journal-prompt-body-bounds' +import { endedJournalReasoning } from './journal-reasoning-row' /** True while an item is still awaiting the row that settles it, so a sink can * treat that row as lifecycle-critical rather than sheddable under pressure. */ @@ -24,6 +26,15 @@ export function requiresTerminalSettlement(body: AgentJournalItemBody): boolean return isRunningAgentJournalTurn(body) } +/** A message still open, ended by a sweep that cannot know when it stopped: no time is claimed. */ +export function endedUnseenMessageBody(body: AgentJournalItemBody): AgentJournalMessageItem | null { + if (body.kind !== 'message' || body.state !== 'running') { + return null + } + const { completedAt: _unseen, ...open } = body + return { ...open, ...endedJournalReasoning() } +} + /** The row that settles an item no one will finish: a running tool call ends as `end` (how its * turn or session ended) says, a pending prompt is cancelled. Null for an item that needs none. * A null `end` ends no call: another writer settled the turn, and its calls stay the provider's. diff --git a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts index 4f21285e417..1bb537a8691 100644 --- a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts +++ b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts @@ -10,6 +10,8 @@ export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:stream_event:content_block_start', 'message:stream_event:content_block_delta', 'message:stream_event:content_block_stop', + // A Messages API keep-alive inside a stream; it carries nothing. + 'message:stream_event:ping', 'message:system:compact_boundary', 'message:system:status', 'message:system:api_retry', diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts index 06c5ba73bdc..141e9ff24bc 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts @@ -38,6 +38,9 @@ describe('provider frame classification catalog', () => { expect(classifyProviderFrame('claude', 'message:system:hook_started', {})).toBe( 'suppressed-benign' ) + expect(classifyProviderFrame('claude', 'message:stream_event:ping', {})).toBe( + 'suppressed-benign' + ) }) it('promotes payload failures over a benign catalog classification', () => { diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts index 56c5244e2ff..da399958eb8 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts @@ -102,6 +102,7 @@ export const PROVIDER_FRAME_CLASSIFICATIONS = { 'message:stream_event:content_block_start': 'status-chrome', 'message:stream_event:content_block_delta': 'stream-into-item', 'message:stream_event:content_block_stop': 'status-chrome', + 'message:stream_event:ping': 'suppressed-benign', 'message:system:compact_boundary': 'status-chrome', 'message:system:status': 'status-chrome', 'message:system:api_retry': 'status-chrome', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-dead-generation-settlement.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-dead-generation-settlement.ts index 25703ea5ba5..3595ff6b0f8 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-dead-generation-settlement.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-dead-generation-settlement.ts @@ -14,6 +14,7 @@ import { readAgentJournalTurn } from '../../../shared/agent-session-turn-record' import { partitionJournalLifecycleMutations } from '../agent-session-journal/journal-lifecycle-batch-partition' import type { JournalLifecycleMutationInput } from '../agent-session-journal/journal-row-builders' import { + endedUnseenMessageBody, requiresTerminalSettlement, runningCallEnd, terminalAgentJournalBody @@ -181,9 +182,10 @@ export async function settleStructuredAgentSessionDeadGeneration(input: { const bodies = new Map(items.map((item) => [item.itemId, item.body])) for (const item of items) { const identity = parseAgentJournalItemKey(item.itemId) - // Ended as its turn is: a proven death cuts a running call short. + // Ended as its turn is: a proven death cuts a running call short. An open reasoning row is + // ended too, but is not unfinished work: its running turn already says so. const end = runningCallEnd(item.turnScope, (id) => bodies.get(id), input.verdict.state) - const body = terminalAgentJournalBody(item.body, end) + const body = endedUnseenMessageBody(item.body) ?? terminalAgentJournalBody(item.body, end) if (identity && body) { mutations.push({ kind: 'item', @@ -246,7 +248,7 @@ export async function settleStaleStructuredAgentSessionState(input: { // A turn already settled (a person's Stop) ends its calls as it ended; only a turn still running // leaves them to the evidence. const end = runningCallEnd(item.turnScope, (id) => journal.itemBody(id), verdictFor(item).state) - const body = terminalAgentJournalBody(item.body, end) + const body = endedUnseenMessageBody(item.body) ?? terminalAgentJournalBody(item.body, end) if (identity && body) { mutations.push({ kind: 'item', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-reasoning-sweep.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-reasoning-sweep.test.ts new file mode 100644 index 00000000000..c54a65a5862 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-reasoning-sweep.test.ts @@ -0,0 +1,110 @@ +// A reasoning row a dead generation left open is ended by both host sweeps, and is not by itself +// evidence that a response was interrupted. +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + AGENT_JOURNAL_THREAD_SCOPE, + type AgentJournalItemBody +} from '../../../shared/agent-session-journal-types' +import { codexProviderHandle } from '../../../shared/agent-session-provider-handle-encoding' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { openTestJournalHostDatabase } from '../agent-session-journal/journal-host-database-test-support' +import { + captureUnfinishedStructuredAgentSessionWork, + settleStaleStructuredAgentSessionState, + settleStructuredAgentSessionDeadGeneration +} from './structured-agent-session-dead-generation-settlement' + +const SESSION = 'session-reasoning-sweep' +const THREAD = 'thread-1' +const REASONING = { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 1 } as const +let root: string +let journal: AgentSessionJournal + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-reasoning-sweep-')) + journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: codexProviderHandle(THREAD) + }, + database: openTestJournalHostDatabase(root), + now: () => 1_000 + }) +}) + +afterEach(async () => { + await journal.close() + await rm(root, { recursive: true, force: true }) +}) + +async function seedOpenReasoning(turnState: 'running' | 'completed'): Promise { + await journal.appendItem( + REASONING, + { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing options' }], + state: 'running' + }, + { fence: 7, turnScope: AGENT_JOURNAL_THREAD_SCOPE } + ) + await journal.appendItem( + { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 2 }, + { kind: 'turn', turnId: 'turn-1', state: turnState, startedAt: 900 }, + { fence: 7, turnScope: AGENT_JOURNAL_THREAD_SCOPE } + ) +} + +function reasoningBody(): AgentJournalItemBody | undefined { + return journal + .snapshot() + .items.find((item) => item.body.kind === 'message' && item.body.role === 'reasoning')?.body +} + +describe('an open reasoning row a dead generation left', () => { + it('ends, with no claimed time, when the generation is settled at its exit', async () => { + await seedOpenReasoning('running') + await expect( + settleStructuredAgentSessionDeadGeneration({ + journal, + sessionId: SESSION, + fence: 8, + settlementId: `expected-close:${SESSION}:8`, + pendingSubmissionReason: 'provider_closed_before_acknowledgement', + verdict: { state: 'interrupted', completedAt: 1_500 }, + showUnexpectedExitOutcome: false + }) + ).resolves.toEqual({ ok: true }) + expect(reasoningBody()).toEqual({ + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing options' }], + state: 'completed' + }) + }) + + it('ends when a later acquisition or reopen sweeps the stale generation', async () => { + await seedOpenReasoning('running') + await settleStaleStructuredAgentSessionState({ + journal, + sessionId: SESSION, + fence: 8, + acquisitionGeneration: 'generation-8', + deathEvidence: null + }) + expect(reasoningBody()).toMatchObject({ state: 'completed' }) + expect(reasoningBody()).not.toHaveProperty('completedAt') + }) + + it('is not by itself unfinished work, so it adds no exit row to a settled turn', async () => { + await seedOpenReasoning('completed') + expect(captureUnfinishedStructuredAgentSessionWork(journal).items).toEqual([]) + }) +}) diff --git a/src/main/runtime/claude-structured-session-integration.test.ts b/src/main/runtime/claude-structured-session-integration.test.ts index bbafd5f1aae..b297404a6f4 100644 --- a/src/main/runtime/claude-structured-session-integration.test.ts +++ b/src/main/runtime/claude-structured-session-integration.test.ts @@ -251,6 +251,10 @@ beforeEach(async () => { resolveShellEnvironmentPolicy: () => shellEnvironmentPolicy, resolveClaudeAuthPolicy: () => claudeAuthPolicy, openClaudeConnection: claude.openConnection, + claudeThinkingDisplay: { + argsFor: async () => ({ 'thinking-display': 'summarized' }), + observeExit: () => {} + }, // Production's sink wiring onto a real hook server, whose records a Stop reaches. statusSink: { publish: (summary, subject) => hookServer.ingestStructuredStatus(summary, subject), @@ -333,6 +337,15 @@ describe('a structured Claude session over agentSession.*', () => { }) }) + // The runtime builds each agent's adapter from a registration; this one must reach Claude's. + it('asks the Claude CLI for readable thinking when the runtime knows it takes the flag', async () => { + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + + expect(claude.live().launch.options.extraArgs).toMatchObject({ + 'thinking-display': 'summarized' + }) + }) + it('passes shell exports straight to the child, as a terminal would', async () => { shellEnv = { ...shellEnv, diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 4333cc7e3a5..f299e17b82e 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -26,6 +26,7 @@ import { nativeChatShellEnvironmentPolicy } from '../../shared/native-chat-shell import { claudeStructuredPermissionModeForSettings } from '../claude/claude-structured-permission-mode' import { codexStructuredPermissionPolicyForSettings } from '../codex/codex-structured-permission-policy' import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' +import { claudeThinkingDisplaySupport } from '../claude/claude-thinking-display-support' export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStartTuiIdleVisibleReadProbe { async getWorktreePs( @@ -154,6 +155,8 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStartTuiIdleVis resolveTuiAgentLaunchEnv('codex', this.requireStore().getSettings().agentDefaultEnv), resolveClaudeLaunchEnv: () => resolveTuiAgentLaunchEnv('claude', this.requireStore().getSettings().agentDefaultEnv), + // Wired only here, so a test runtime never runs a real `claude --version`. + claudeThinkingDisplay: claudeThinkingDisplaySupport, resolveShellEnvironmentPolicy: () => nativeChatShellEnvironmentPolicy(this.requireStore().getSettings()), resolveClaudeAuthPolicy: () => diff --git a/src/main/runtime/structured-agent-runtime-registrations.ts b/src/main/runtime/structured-agent-runtime-registrations.ts index 5db941bb8d8..bfe4e218893 100644 --- a/src/main/runtime/structured-agent-runtime-registrations.ts +++ b/src/main/runtime/structured-agent-runtime-registrations.ts @@ -115,6 +115,7 @@ function createClaudeAdapter( store, resolveWorkspacePath: deps.resolveWorkspacePath, ...(deps.resolveClaudeCommand ? { resolveClaudeCommand: deps.resolveClaudeCommand } : {}), + ...(deps.claudeThinkingDisplay ? { claudeThinkingDisplay: deps.claudeThinkingDisplay } : {}), ...(deps.resolveClaudeLaunchEnv ? { resolveClaudeLaunchEnv: deps.resolveClaudeLaunchEnv } : {}), resolveClaudeInheritedEnv: context.environment.resolveClaudeInheritedEnv, resolveClaudeAuthPolicy: deps.resolveClaudeAuthPolicy, diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index 88544a6c806..c347be27270 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -59,6 +59,7 @@ import { modelCatalogHostDeps, type RuntimeAgentAccountHomeResolver } from './structured-agent-model-catalog-wiring' +import type { ClaudeThinkingDisplaySupport } from '../claude/claude-thinking-display-support' /** Whether this profile holds a structured chat: a record or tab in the journal database, or the * records file a profile from before it carries while the database still owes its copy. */ @@ -93,6 +94,8 @@ export type StructuredAgentSessionRuntimeDeps = { resolveWorkspacePath: (workspaceId: string) => Promise resolveCodexCommand?: (options?: { pathEnv?: string | null; homePath?: string }) => string resolveClaudeCommand?: () => string + /** Whether a Claude CLI takes the thinking-display flag; absent never passes it. */ + claudeThinkingDisplay?: ClaudeThinkingDisplaySupport /** Provider transports are overridden only to drive the runtime against scripted children. */ openCodexConnection?: CodexStructuredSessionAdapterDeps['openConnection'] openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] diff --git a/src/main/runtime/structured-claude-runtime-adapter.ts b/src/main/runtime/structured-claude-runtime-adapter.ts index 8b4b70f8bc2..cb25f94db7c 100644 --- a/src/main/runtime/structured-claude-runtime-adapter.ts +++ b/src/main/runtime/structured-claude-runtime-adapter.ts @@ -18,11 +18,15 @@ import { import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' import type { AgentSessionRecordStore } from './agent-session-record-store' import { ClaudeAtRestCommandCatalog } from '../claude/claude-at-rest-commands' +import { openClaudeStreamJsonConnection } from '../claude/claude-stream-json-connection' +import type { ClaudeThinkingDisplaySupport } from '../claude/claude-thinking-display-support' export type StructuredClaudeRuntimeAdapterDeps = { store: AgentSessionRecordStore resolveWorkspacePath: (workspaceId: string) => Promise resolveClaudeCommand?: () => string + /** Whether a Claude CLI takes the thinking-display flag; absent never passes it. */ + claudeThinkingDisplay?: ClaudeThinkingDisplaySupport resolveClaudeLaunchEnv?: () => Promise> | Record /** The env a Claude child inherits before auth stripping; absent inherits Orca's own. */ resolveClaudeInheritedEnv?: () => Promise> @@ -94,7 +98,8 @@ export function createStructuredClaudeRuntimeAdapter( : {}), ...(deps.readClaudeManagedAccountGate ? { readManagedAccountGate: deps.readClaudeManagedAccountGate } - : {}) + : {}), + ...(deps.claudeThinkingDisplay ? { thinkingDisplay: deps.claudeThinkingDisplay } : {}) }), persistHandle: async ({ sessionId, providerSessionId, leafUuid, fence }) => { const currentFence = store.getRecord(sessionId)?.lease.runtimeFence ?? fence @@ -134,8 +139,38 @@ export function createStructuredClaudeRuntimeAdapter( ...(deps.onSessionIdle ? { onSessionIdle: deps.onSessionIdle } : {}), ...(deps.onChildWorkEvidence ? { onChildWorkEvidence: deps.onChildWorkEvidence } : {}), ...(deps.logger ? { logger: deps.logger } : {}), - ...(deps.openClaudeConnection ? { openConnection: deps.openClaudeConnection } : {}), + ...openClaudeConnectionOf(deps), ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}), ...(deps.modelCatalog ? { modelCatalog: deps.modelCatalog } : {}) }) } + +/** The child's connection, watched for a CLI refusing the thinking-display flag, so the start that + * failed on it is the last one to pass it. */ +export function openClaudeConnectionOf( + deps: Pick +): Pick { + const support = deps.claudeThinkingDisplay + if (!support) { + return deps.openClaudeConnection ? { openConnection: deps.openClaudeConnection } : {} + } + const open = deps.openClaudeConnection ?? openClaudeStreamJsonConnection + return { + openConnection: (launch, handlers = {}, ...rest) => + open( + launch, + { + ...handlers, + // Every argument passes through, so one the connection adds later still reaches the session. + onExit: (...args) => { + support.observeExit( + { command: launch.pathToClaudeCodeExecutable, cwd: launch.cwd }, + args[0] + ) + handlers.onExit?.(...args) + } + }, + ...rest + ) + } +} diff --git a/src/main/runtime/structured-claude-thinking-display-refusal.test.ts b/src/main/runtime/structured-claude-thinking-display-refusal.test.ts new file mode 100644 index 00000000000..817add96e67 --- /dev/null +++ b/src/main/runtime/structured-claude-thinking-display-refusal.test.ts @@ -0,0 +1,97 @@ +// A real child, through the real SDK connection, refusing the thinking-display flag the way an +// older Claude CLI does: commander's message and exit code 1. +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from '../claude/claude-structured-launch-resolution' +import { fakeClaude } from '../claude/claude-structured-session-test-support' +import { createClaudeThinkingDisplaySupport } from '../claude/claude-thinking-display-support' +import { openClaudeConnectionOf } from './structured-claude-runtime-adapter' + +const FAKE_CLI = join( + __dirname, + '..', + 'claude', + '__fixtures__', + 'claude-agent-sdk-scripted-cli.mjs' +) +const scratchDirs: string[] = [] + +afterEach(() => { + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +async function startRefusing(stderr: string) { + const dir = mkdtempSync(join(tmpdir(), 'claude-thinking-display-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + writeFileSync(scenarioPath, JSON.stringify({ steps: [{ stderr }, { exit: 1 }] })) + const env = { PATH: process.env.PATH ?? '', ORCA_SDK_CONTRACT_SCENARIO_PATH: scenarioPath } + const probe = vi.fn(async () => '2.1.280') + const support = createClaudeThinkingDisplaySupport({ + probe, + keyOf: async (command, cwd) => `${command}\n${cwd}`, + budgetMs: 1_000, + now: () => performance.now() + }) + const flag = await support.argsFor({ command: FAKE_CLI, cwd: dir, env }) + const { openConnection } = openClaudeConnectionOf({ claudeThinkingDisplay: support }) + let exited: Error | null = null + const connection = await openConnection!( + { + pathToClaudeCodeExecutable: FAKE_CLI, + options: { + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs, ...flag } + }, + cwd: dir, + env + }, + { onExit: (error) => (exited = error) } + ) + await vi.waitFor(() => expect(exited).not.toBeNull(), { timeout: 10_000 }) + await connection.close() + return { support, probe, flag, launch: { command: FAKE_CLI, cwd: dir, env } } +} + +describe('a Claude CLI that refuses the thinking-display flag', () => { + it('fails that one start as today, and the next launch skips the flag', async () => { + const { support, probe, flag, launch } = await startRefusing( + "error: unknown option '--thinking-display'\n" + ) + expect(flag).toEqual({ 'thinking-display': 'summarized' }) + await vi.waitFor(async () => expect(await support.argsFor(launch)).toEqual({})) + expect(probe).toHaveBeenCalledTimes(1) + }) + + it('hands the session whether its close was the one Orca began', async () => { + const claude = fakeClaude() + const support = createClaudeThinkingDisplaySupport({ + probe: async () => '2.1.280', + keyOf: async () => null, + budgetMs: 1_000, + now: () => performance.now() + }) + const { openConnection } = openClaudeConnectionOf({ + claudeThinkingDisplay: support, + openClaudeConnection: claude.openConnection + }) + const onExit = vi.fn() + await openConnection!( + { pathToClaudeCodeExecutable: FAKE_CLI, options: {}, cwd: '/w' }, + { onExit } + ) + const exited = new Error('closed') + claude.connections[0]!.handlers.onExit?.(exited, { expected: true }) + // Without it, a Stop's own exit would read as the child exiting on its own. + expect(onExit).toHaveBeenCalledWith(exited, { expected: true }) + }) + + it('records nothing when the start failed for another reason', async () => { + const { support, launch } = await startRefusing('claude: not signed in\n') + await expect(support.argsFor(launch)).resolves.toEqual({ 'thinking-display': 'summarized' }) + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.live-reasoning.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.live-reasoning.test.tsx new file mode 100644 index 00000000000..505c87d763e --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.live-reasoning.test.tsx @@ -0,0 +1,308 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' +import type { NativeChatLiveSession } from './use-native-chat-live-session' +import { NativeChatMessageList } from './NativeChatMessageList' +import { installNativeChatMessageListTestViewport } from './native-chat-message-list-test-viewport' +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' + +let restoreViewport = (): void => {} +beforeAll(() => { + restoreViewport = installNativeChatMessageListTestViewport() +}) +afterAll(() => restoreViewport()) +afterEach(cleanup) + +const STARTED = 1_000 + +const prompt: NativeChatMessage = { + id: 'user-1', + role: 'user', + blocks: [{ type: 'text', text: 'Start the task' }], + timestamp: STARTED - 500, + source: 'transcript' +} + +function reasoning( + id: string, + text: string, + state: 'running' | 'completed', + fields: Partial = {} +): NativeChatMessage { + return { + id, + role: 'reasoning', + blocks: [{ type: 'text', text }], + timestamp: STARTED, + source: 'transcript', + state, + ...(state === 'completed' ? { completedAt: STARTED + 12_000 } : {}), + ...fields + } +} + +/** The journal that says the turn runs and what its newest content is. */ +function journal(rows: readonly NativeChatMessage[]): AgentJournalRenderItem[] { + return [ + { + itemId: prompt.id, + revision: 1, + sequence: 1, + observedAt: 1, + body: { kind: 'message', role: 'user', blocks: prompt.blocks } + }, + { + itemId: 'turn-1', + revision: 1, + sequence: 2, + observedAt: 2, + body: { kind: 'turn', turnId: 'turn-1', state: 'running', userItemId: prompt.id } + }, + ...rows.map((row, index) => ({ + itemId: row.id, + revision: 1, + sequence: index + 3, + observedAt: index + 3, + ...(row.agentId ? { agentId: row.agentId } : {}), + body: { + kind: 'message' as const, + role: row.role, + blocks: row.blocks, + ...(row.state ? { state: row.state } : {}) + } + })) + ] +} + +function sessionOf(messages: NativeChatMessage[]): NativeChatLiveSession { + return { + messages, + status: 'working', + sessionId: 'session-1', + agent: 'claude', + hasMore: false, + loadingEarlier: false, + olderHistoryGeneration: 0, + loadEarlier: vi.fn(), + readPhase: 'ready' + } +} + +function list( + rows: readonly NativeChatMessage[], + props: Partial> = {} +): React.JSX.Element { + return ( + + ) +} + +const liveLine = (): HTMLElement => + screen.getByText('Thinking').closest('[data-native-chat-turn-activity]')! + +describe('live reasoning, read through the one live line', () => { + it('shows one "Thinking", collapsed, and no row for the open block', () => { + render(list([reasoning('r-1', 'Weighing two approaches', 'running')])) + expect(screen.getAllByText('Thinking')).toHaveLength(1) + const toggle = screen.getByRole('button', { name: 'Thinking' }) + expect(toggle).toHaveAttribute('aria-expanded', 'false') + expect(liveLine()).toContainElement(toggle) + expect(screen.queryByRole('button', { name: /Reasoning|Thought/ })).toBeNull() + expect(screen.queryByText('Weighing two approaches')).toBeNull() + }) + + it('opens to the live text, which follows the block as it grows', () => { + const { rerender } = render(list([reasoning('r-1', 'Weighing two approaches', 'running')])) + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + expect(screen.getByRole('button', { name: 'Thinking' })).toHaveAttribute( + 'aria-expanded', + 'true' + ) + expect(screen.getByText('Weighing two approaches')).toBeInTheDocument() + rerender(list([reasoning('r-1', 'Weighing two approaches, then the cheaper one', 'running')])) + expect(screen.getByText('Weighing two approaches, then the cheaper one')).toBeInTheDocument() + }) + + it('keeps the body out of the live region, which announces the label only', () => { + render(list([reasoning('r-1', 'Weighing two approaches', 'running')])) + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + const body = screen.getByText('Weighing two approaches') + expect(body.closest('[aria-live]')).toBeNull() + expect(screen.getByText('Thinking').closest('[aria-live]')).not.toBeNull() + }) + + it('lands the finished row open when the reader opened it live, while the turn works on', () => { + const { rerender } = render(list([reasoning('r-1', 'Weighing two approaches', 'running')])) + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + rerender(list([reasoning('r-1', 'Weighing two approaches', 'completed')])) + // The line no longer discloses anything; the row does, still open. + expect(screen.queryByRole('button', { name: /Thinking|Working/ })).toBeNull() + expect(screen.getByRole('button', { name: 'Reasoning: Thought for 12s' })).toHaveAttribute( + 'aria-expanded', + 'true' + ) + expect(screen.getByText('Weighing two approaches')).toBeInTheDocument() + }) + + // The body owns its colour: inherited, it read full foreground under the line and muted in the row, + // so the text dimmed as the block landed. + it('draws the same body, in its own quieter tone, live and once landed', () => { + const body = () => + screen.getByText('Weighing two approaches').closest('[data-native-chat-message-tone]') + const { rerender } = render(list([reasoning('r-1', 'Weighing two approaches', 'running')])) + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + const live = body() + expect(live).toHaveAttribute('data-native-chat-message-tone', 'faint') + expect(live).toHaveClass('text-chat-foreground-faint', 'pl-5.5', 'max-h-80') + const liveClasses = live?.getAttribute('class') + rerender(list([reasoning('r-1', 'Weighing two approaches', 'completed')])) + expect(body()?.getAttribute('class')).toBe(liveClasses) + }) + + it('starts the next block collapsed', () => { + const { rerender } = render(list([reasoning('r-1', 'First thought', 'running')])) + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + rerender( + list([reasoning('r-1', 'First thought', 'completed'), reasoning('r-2', 'Second', 'running')]) + ) + expect(screen.getByRole('button', { name: 'Thinking' })).toHaveAttribute( + 'aria-expanded', + 'false' + ) + expect(screen.queryByText('Second')).toBeNull() + }) + + it('is not expandable while the open block has no text yet', () => { + render(list([reasoning('r-1', '', 'running')])) + expect(screen.getAllByText('Thinking')).toHaveLength(1) + expect(screen.queryByRole('button', { name: 'Thinking' })).toBeNull() + }) + + it('draws the open row when a waiting prompt replaces the line', () => { + render( + list([reasoning('r-1', 'Weighing two approaches', 'running')], { + awaitingInput: 'unshown' + }) + ) + expect(screen.queryByText('Thinking')).toBeNull() + expect(screen.getByRole('button', { name: 'Reasoning' })).toHaveAttribute( + 'aria-expanded', + 'false' + ) + }) + + // The open block can sit on a slot kept for its turn's bar or diff rollup; only the slot stays. + it('draws no row for the open block under a diff rollup on its turn', () => { + const edit: NativeChatMessage = { + id: 'edit-1', + role: 'assistant', + blocks: [ + { type: 'tool-call', name: 'Diff', input: { path: 'a.ts' }, state: 'completed' }, + { type: 'tool-result', output: '@@ -1 +1 @@\n-old\n+new' } + ], + timestamp: STARTED, + source: 'transcript' + } + render( + list([ + edit, + reasoning('r-1', 'Weighing two approaches', 'running', { timestamp: STARTED + 50 }) + ]) + ) + expect(screen.queryByRole('button', { name: /Reasoning/ })).toBeNull() + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + expect(screen.getAllByText('Weighing two approaches')).toHaveLength(1) + }) + + it('draws no row for the open block that carries a provider-opened turn bar', () => { + const asked: NativeChatMessage = { ...prompt, id: 'u1', timestamp: 1 } + const done: NativeChatMessage = { + id: 'a1', + role: 'assistant', + blocks: [{ type: 'text', text: 'Done.' }], + timestamp: 2, + source: 'transcript' + } + const woke = reasoning('r-w', 'Checking the background build', 'running', { timestamp: 3 }) + const thread = { kind: 'thread' as const } + const inTurn = (turnItemId: string) => ({ kind: 'turn' as const, turnItemId }) + const item = ( + itemId: string, + sequence: number, + body: AgentJournalRenderItem['body'], + turnScope: AgentJournalRenderItem['turnScope'] + ): AgentJournalRenderItem => ({ + itemId, + revision: 0, + sequence, + observedAt: sequence, + turnScope, + body + }) + const items = [ + item('u1', 1, { kind: 'message', role: 'user', blocks: asked.blocks }, thread), + item('t1', 2, { kind: 'turn', turnId: 't1', state: 'completed', userItemId: 'u1' }, thread), + item('a1', 3, { kind: 'message', role: 'assistant', blocks: done.blocks }, inTurn('t1')), + item( + 'wake', + 4, + { kind: 'turn', turnId: 'wake', state: 'running', userItemId: 'claude:wake' }, + thread + ), + item( + 'r-w', + 5, + { kind: 'message', role: 'reasoning', blocks: woke.blocks, state: 'running' }, + inTurn('wake') + ) + ] + render( + list([], { + session: sessionOf([asked, done, woke]), + journalItems: items + }) + ) + expect(screen.getByText(/Working for/)).toBeInTheDocument() + expect(screen.queryByRole('button', { name: /Reasoning/ })).toBeNull() + fireEvent.click(screen.getByRole('button', { name: 'Thinking' })) + expect(screen.getAllByText('Checking the background build')).toHaveLength(1) + }) + + // A live region only announces changes to itself; a replaced one says nothing. + it('keeps one live region while the line turns into the disclosure and back', () => { + const tool: NativeChatMessage = { + id: 'tool-1', + role: 'assistant', + blocks: [{ type: 'tool-call', name: 'shell', input: { command: 'ls' }, state: 'completed' }], + timestamp: STARTED, + source: 'transcript' + } + const later = { timestamp: STARTED + 50 } + const region = () => document.querySelector('[data-native-chat-turn-activity][aria-live]') + const { rerender } = render(list([tool])) + const before = region() + expect(before).toHaveTextContent('Working…') + rerender(list([tool, reasoning('r-1', 'Weighing two approaches', 'running', later)])) + expect(region()).toBe(before) + expect(before).toHaveTextContent('Thinking') + rerender( + list([ + tool, + reasoning('r-1', 'Weighing two approaches', 'completed', later), + { ...tool, id: 'tool-2', timestamp: STARTED + 60 } + ]) + ) + expect(region()).toBe(before) + expect(before).toHaveTextContent('Working…') + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx index 5e406876a56..471f6fd3470 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx @@ -29,10 +29,10 @@ import { import type { NativeChatTranscriptRowContext } from './NativeChatTranscriptRow' import type { NativeChatDeliveryNotice } from './NativeChatMessageRow' import { - buildNativeChatTranscriptSlots, splitNativeChatSlotsWaitingBehindLiveTurn, nativeChatSlotIndexOf } from './native-chat-transcript-slots' +import { useNativeChatTranscriptSlots } from './use-native-chat-transcript-slots' import { useNativeChatTranscriptWindow } from './use-native-chat-transcript-window' import { nativeChatRowsInTranscriptOrder } from './native-chat-subagent-sections' import { useNativeChatSubagentSections } from './use-native-chat-subagent-sections' @@ -187,37 +187,25 @@ export function NativeChatMessageList({ : null const lifecycleWorking = session.transcriptLifecycle?.state === 'working' const { measureContent, typography } = useNativeChatRowTypography(contentRef) - const allSlots = useMemo( - () => - buildNativeChatTranscriptSlots({ - typography, - messages: rows, - turnKeys, - liveTurnKey, - receipts, - turnStatuses, - turnDiffs, - expandedTurnKeys: expandedTurnIds, - isWorking, - lifecycleWorking, - subagentSections, - subagentChoices - }), - [ - typography, - liveTurnKey, - expandedTurnIds, - isWorking, - lifecycleWorking, - receipts, - rows, - subagentChoices, - subagentSections, - turnDiffs, - turnKeys, - turnStatuses - ] - ) + const { slots: allSlots, liveLine } = useNativeChatTranscriptSlots({ + typography, + messages: rows, + turnKeys, + liveTurnKey, + receipts, + turnStatuses, + turnDiffs, + expandedTurnKeys: expandedTurnIds, + isWorking, + lifecycleWorking, + subagentSections, + subagentChoices, + line: { + draws: tailRow === 'activity', + thinking: turnStatuses.active?.thinking === true, + activityText: turnActivity?.text + } + }) // A message waiting behind the live turn draws after that turn's live activity, not inside it. const { slots, waitingSlots } = useMemo( () => splitNativeChatSlotsWaitingBehindLiveTurn(allSlots, journalItems), @@ -396,10 +384,11 @@ export function NativeChatMessageList({ context={rowContext} window={transcriptWindow} /> - {tailRow === 'activity' ? ( + {liveLine ? ( ) : tailRow === 'awaiting-input' ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatMessageRow.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageRow.test.tsx index 01c5b890b9c..a7e8ef689bc 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageRow.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageRow.test.tsx @@ -115,11 +115,16 @@ describe('MessageRow control visibility', () => { expect(screen.queryAllByRole('button')).toHaveLength(role === 'assistant' ? 2 : 1) }) - it.each(['reasoning', 'system'] as const)('preserves chrome-free %s rows', (role) => { - renderMessage(role) - expect(screen.queryByRole('time')).toBeNull() - expect(screen.queryByRole('button')).toBeNull() - }) + it.each(['reasoning', 'system'] as const)( + 'omits timestamp and agent controls on %s rows', + (role) => { + renderMessage(role) + expect(screen.queryByRole('time')).toBeNull() + expect(screen.queryByRole('button', { name: 'Copy message' })).toBeNull() + expect(screen.queryByRole('button', { name: 'Scroll this message to top' })).toBeNull() + expect(screen.queryAllByRole('button')).toHaveLength(role === 'reasoning' ? 1 : 0) + } + ) }) describe('MessageRow send mode', () => { diff --git a/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx b/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx index b6c24123117..a19d41ad2a5 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx @@ -13,6 +13,7 @@ import type { } from '../../../../shared/native-chat-types' import { deriveNativeChatRowContent } from '../../../../shared/native-chat-row-content' import { NativeChatToolRun } from './NativeChatToolRun' +import { NativeChatReasoningRow } from './NativeChatReasoningRow' import { NativeChatCodeBlock } from './NativeChatCodeBlock' import { NativeChatNoticeRow } from './NativeChatNoticeRow' import { NativeChatCopyButton } from './NativeChatCopyButton' @@ -92,7 +93,6 @@ export const MessageRow = memo(function MessageRow({ onLinkClick, allowFileUriLinks = false, deliveryNotice, - folded = false, subagentRoster, subagentDisclosure, inSubagentSection = false, @@ -112,8 +112,6 @@ export const MessageRow = memo(function MessageRow({ onLinkClick?: CommentMarkdownLinkClickHandler allowFileUriLinks?: boolean deliveryNotice?: NativeChatDeliveryNotice - /** Behind a folded turn: the row keeps only what outlives the turn. */ - folded?: boolean /** On a roster row: its list's state and the subagents whose rows open below it. */ subagentRoster?: NativeChatSubagentRosterState subagentDisclosure?: NativeChatSubagentDisclosure @@ -152,12 +150,6 @@ export const MessageRow = memo(function MessageRow({ return null } - // Behind a folded turn this row is the work, not the answer. Rows that outlive - // their turn never reach here — the fold leaves them out. - if (folded) { - return null - } - const notice = isSystem ? message.blocks.find( (block) => @@ -248,9 +240,22 @@ export const MessageRow = memo(function MessageRow({ ) } - // Plain assistant prose is the copyable unit; reasoning/system asides stay - // chrome-free. Controls reveal on hover/keyboard focus and stay visible on touch. - const showControls = !isReasoning && !isSystem && markdown.length > 0 + if (isReasoning) { + return ( +
+ +
+ ) + } + + // Assistant controls reveal on hover and keyboard focus; system asides stay chrome-free. + const showControls = !isSystem && markdown.length > 0 return (
+ +
+ ) +} + +/** The disclosure caret of a `group/reasoning` header: shown on hover, keyboard focus (on the + * header or a trigger inside it) and touch, and turned while open. */ +export function NativeChatReasoningChevron(): React.JSX.Element { + return ( + + ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatReasoningRow.test.tsx b/src/renderer/src/components/native-chat/NativeChatReasoningRow.test.tsx new file mode 100644 index 00000000000..05ce6c8807a --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatReasoningRow.test.tsx @@ -0,0 +1,217 @@ +// @vitest-environment happy-dom +import '@testing-library/jest-dom/vitest' +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import { NativeChatReasoningRow } from './NativeChatReasoningRow' +import { MessageRow } from './NativeChatMessageRow' +import { NativeChatToolRunIcon } from './NativeChatToolIcon' +import { + NativeChatDisclosureContext, + useNativeChatDisclosures +} from './native-chat-disclosure-store' + +vi.mock('@/components/sidebar/CommentMarkdown', () => ({ + default: ({ content }: { content: string }) =>
{content}
+})) + +afterEach(() => { + cleanup() + vi.restoreAllMocks() +}) + +const STARTED = 100_000 + +describe('reasoning disclosure', () => { + it('starts collapsed without mounting markdown', () => { + render( + + ) + expect(screen.getByRole('button', { name: 'Reasoning: Thought' })).toHaveAttribute( + 'aria-expanded', + 'false' + ) + expect(screen.queryByTestId('markdown')).not.toBeInTheDocument() + // The same category glyph slot a tool row leads with, hidden from the accessible name. + const glyph = screen.getByRole('button').querySelector('svg.lucide-brain') + expect(glyph).toHaveAttribute('aria-hidden', 'true') + }) + + it('leads with the shared vocabulary brain, drawn exactly as a tool row draws a glyph', () => { + const { container } = render( + + ) + const shared = container.innerHTML + cleanup() + render( + + ) + expect( + screen.getByRole('button').querySelector('svg.lucide-brain')?.parentElement?.outerHTML + ).toBe(shared) + }) + + it('hides its chevron only where hover can reveal it, and shows it on keyboard focus and once open', () => { + render( + + ) + // An SVG's `className` is an `SVGAnimatedString`, so read the attribute. + const chevron = screen.getByRole('button').querySelector('svg.lucide-chevron-right') + const classes = (chevron?.getAttribute('class') ?? '').split(' ') + // Touch has no hover, so an ungated `opacity-0` would hide it there for good. + expect(classes).not.toContain('opacity-0') + expect(classes).toEqual( + expect.arrayContaining([ + 'can-hover:opacity-0', + 'group-hover/reasoning:opacity-100', + 'group-focus-visible/reasoning:opacity-100', + 'group-data-[state=open]/reasoning:opacity-100' + ]) + ) + }) + + it('expands through a native button and keeps disclosure state through revisions', () => { + const message = { + id: 'r-1', + role: 'reasoning' as const, + timestamp: STARTED, + state: 'completed' as const + } + const { rerender } = render() + const trigger = screen.getByRole('button') + expect(trigger.tagName).toBe('BUTTON') + fireEvent.click(trigger) + expect(trigger).toHaveAttribute('aria-expanded', 'true') + expect(screen.getByTestId('markdown')).toHaveTextContent('Inspecting') + rerender() + expect(screen.getByRole('button')).toBe(trigger) + expect(trigger).toHaveAttribute('aria-expanded', 'true') + expect(screen.getByTestId('markdown')).toHaveTextContent('More') + // Collapsed again by the user, it stays collapsed through the next revision. + fireEvent.click(trigger) + rerender() + expect(trigger).toHaveAttribute('aria-expanded', 'false') + expect(screen.queryByTestId('markdown')).not.toBeInTheDocument() + }) + + it('keeps its disclosure under the block key, so a remount (or the live line) finds it open', () => { + const Transcript = ({ children }: { children: React.ReactNode }) => { + const disclosures = useNativeChatDisclosures() + return ( + + {children} + + ) + } + const row = ( + + ) + const { rerender } = render({row}) + fireEvent.click(screen.getByRole('button')) + // Windowed out, then back. + rerender({null}) + rerender({row}) + expect(screen.getByRole('button')).toHaveAttribute('aria-expanded', 'true') + expect(screen.getByTestId('markdown')).toHaveTextContent('Inspecting') + }) + + it.each(['', ' \n\t'])('draws nothing for blank reasoning %j', (markdown) => { + const { container } = render( + + ) + expect(container).toBeEmptyDOMElement() + }) +}) + +describe('the reasoning headline', () => { + const headline = ( + message: Pick, + turnIsWorking = false + ) => { + render( + + ) + return screen.queryByRole('button')?.textContent ?? null + } + + // The live line hides the one block it discloses; any other open row draws, claiming no end. + it('reads Reasoning while the row is open and its turn or subagent is running', () => { + expect(headline({ timestamp: STARTED, state: 'running' }, true)).toBe('Reasoning') + }) + + it('reads Thought for N s once it closes, while the turn goes on working', () => { + expect( + headline({ timestamp: STARTED, state: 'completed', completedAt: STARTED + 12_000 }, true) + ).toContain('Thought for 12s') + }) + + it('measures the span the host saw, at least one second', () => { + expect( + headline({ timestamp: STARTED, state: 'completed', completedAt: STARTED + 65_000 }) + ).toContain('Thought for 1m 5s') + cleanup() + expect( + headline({ timestamp: STARTED, state: 'completed', completedAt: STARTED + 300 }) + ).toContain('Thought for 1s') + }) + + it('claims no duration it never saw, and draws an open row in a settled turn as Thought', () => { + expect(headline({ timestamp: STARTED, state: 'completed' })).toBe('Reasoning: Thought') + cleanup() + expect(headline({ timestamp: STARTED, state: 'running' })).toBe('Reasoning: Thought') + }) + + it('stays neutral for a row from a host that kept no lifecycle', () => { + expect(headline({ timestamp: STARTED }, true)).toBe('Reasoning') + }) + + it('draws through the message row while open, and reads its span once it closes', () => { + const message: NativeChatMessage = { + id: 'reasoning-1', + role: 'reasoning', + source: 'transcript', + timestamp: STARTED, + state: 'running', + blocks: [{ type: 'text', text: 'Inspecting the request\nFull reasoning' }] + } + const { rerender } = render( + + ) + expect(screen.getByRole('button')).toHaveTextContent('Reasoning') + rerender( + + ) + expect(screen.getByRole('button')).toHaveTextContent('Thought for 3s') + fireEvent.click(screen.getByRole('button')) + expect(screen.getByTestId('markdown')).toHaveTextContent('Full reasoning') + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatReasoningRow.tsx b/src/renderer/src/components/native-chat/NativeChatReasoningRow.tsx new file mode 100644 index 00000000000..040bd6c6206 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatReasoningRow.tsx @@ -0,0 +1,77 @@ +import type { CommentMarkdownLinkClickHandler } from '@/components/sidebar/CommentMarkdown' +import { Collapsible, CollapsibleContent, CollapsibleTrigger } from '@/components/ui/collapsible' +import { translate } from '@/i18n/i18n' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import { + nativeChatReasoningDisclosureKey, + nativeChatReasoningHeadline, + type NativeChatReasoningHeadline +} from '../../../../shared/native-chat-reasoning-row' +import { + NativeChatReasoningBody, + NativeChatReasoningChevron +} from './NativeChatReasoningDisclosure' +import { NativeChatToolRunIcon } from './NativeChatToolIcon' +import { useNativeChatDisclosure } from './native-chat-disclosure-store' + +function translatedHeadline(headline: NativeChatReasoningHeadline): string { + if (headline.kind === 'thoughtFor') { + return translate('components.native-chat.thoughtForDuration', 'Thought for {{duration}}', { + duration: headline.duration + }) + } + return headline.kind === 'thought' + ? translate('components.native-chat.thought', 'Thought') + : translate('components.native-chat.reasoning', 'Reasoning') +} + +export function NativeChatReasoningRow({ + message, + markdown, + turnIsWorking = false, + onLinkClick, + allowFileUriLinks +}: { + message: Pick + markdown: string + /** The row's own turn (or subagent) is still running; an open row there has not ended. */ + turnIsWorking?: boolean + onLinkClick?: CommentMarkdownLinkClickHandler + allowFileUriLinks?: boolean +}): React.JSX.Element | null { + // Keyed like the live line, so a block opened while it streamed lands open, and windowing keeps it. + const disclosure = useNativeChatDisclosure(nativeChatReasoningDisclosureKey(message.id), false) + if (!markdown.trim()) { + return null + } + const label = translate('components.native-chat.reasoning', 'Reasoning') + const headline = translatedHeadline(nativeChatReasoningHeadline(message, { live: turnIsWorking })) + + return ( +
+ + + {/* Laid out like a tool run's header, so its glyph sits in the same column. */} + + + + + + +
+ ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatToolIcon.tsx b/src/renderer/src/components/native-chat/NativeChatToolIcon.tsx index 11bde34c6ce..01f92299000 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolIcon.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolIcon.tsx @@ -1,5 +1,6 @@ import { Bot, + Brain, Eye, Folder, Globe, @@ -29,7 +30,8 @@ const NATIVE_CHAT_TOOL_GLYPHS: Record = { bot: Bot, 'list-checks': ListChecks, wrench: Wrench, - 'message-square-more': MessageSquareMore + 'message-square-more': MessageSquareMore, + brain: Brain } /** The fixed 16px slot with a 14px glyph, which keeps every row left-aligned diff --git a/src/renderer/src/components/native-chat/NativeChatTranscriptRow.tsx b/src/renderer/src/components/native-chat/NativeChatTranscriptRow.tsx index bcb3841cbb9..f1464b11b7d 100644 --- a/src/renderer/src/components/native-chat/NativeChatTranscriptRow.tsx +++ b/src/renderer/src/components/native-chat/NativeChatTranscriptRow.tsx @@ -95,7 +95,7 @@ export const NativeChatTranscriptRow = memo(function NativeChatTranscriptRow({
{/* A turn with no user bubble carries its bar above its first row. */} {slot.statusAbove ? statusRow : null} - {receipt ? ( + {!slot.drawsMessage ? null : receipt ? ( ) : ( 0} diff --git a/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx b/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx index 4d59d159f25..4a3d62a2f8f 100644 --- a/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx +++ b/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx @@ -1,35 +1,83 @@ +import { useId } from 'react' import { Loader2 } from 'lucide-react' +import type { CommentMarkdownLinkClickHandler } from '@/components/sidebar/CommentMarkdown' +import { Collapsible, CollapsibleContent, CollapsibleTrigger } from '@/components/ui/collapsible' import { translate } from '@/i18n/i18n' -import type { NativeChatTurnActivity } from '../../../../shared/native-chat-turn-activity' +import type { NativeChatLiveLine } from '../../../../shared/native-chat-live-line' +import { nativeChatReasoningDisclosureKey } from '../../../../shared/native-chat-reasoning-row' import { describeNativeChatActiveTurnLabel } from '../../../../shared/native-chat-turn-status' +import { + NativeChatReasoningBody, + NativeChatReasoningChevron +} from './NativeChatReasoningDisclosure' +import { useNativeChatDisclosure } from './native-chat-disclosure-store' /** The live turn's tail line: a spinner plus what the turn is doing right now — * the provider's activity text, else that it is reasoning, else plain "Working…". - * The clock lives in the turn bar under the user's message, not here. */ + * The clock lives in the turn bar under the user's message, not here. While the + * agent's open reasoning block has text, the line is that block's disclosure. */ export function NativeChatTurnActivityLine({ - activity, - thinking + line, + onLinkClick, + allowFileUriLinks }: { - activity?: NativeChatTurnActivity | null - thinking: boolean + line: NativeChatLiveLine + onLinkClick?: CommentMarkdownLinkClickHandler + allowFileUriLinks?: boolean }): React.JSX.Element { - const resolved = describeNativeChatActiveTurnLabel({ activityText: activity?.text, thinking }) + const resolved = describeNativeChatActiveTurnLabel(line) const label = resolved.source === 'activity' ? resolved.text : resolved.key === 'thinking' ? translate('components.native-chat.status.thinking', 'Thinking') : translate('components.native-chat.status.working', 'Working…') + const { reasoning } = line + // The finished row reads this key too, so a block opened here lands open once it ends. + const disclosure = useNativeChatDisclosure( + reasoning ? nativeChatReasoningDisclosureKey(reasoning.message.id) : undefined, + false + ) + const open = reasoning !== null && disclosure.open + const labelId = useId() return ( -
- - {label} -
+ + {/* One element for every state of the line, so a screen reader hears each new label. The + trigger overlays it, rather than wrapping it, and the body sits outside it. */} +
+ + + {label} + + {reasoning ? ( + <> + + +
+ {reasoning ? ( + + + + ) : null} +
) } diff --git a/src/renderer/src/components/native-chat/native-chat-row-height-estimate.test.ts b/src/renderer/src/components/native-chat/native-chat-row-height-estimate.test.ts index 9b7d2164405..67767958732 100644 --- a/src/renderer/src/components/native-chat/native-chat-row-height-estimate.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-row-height-estimate.test.ts @@ -172,6 +172,18 @@ describe('transcript row height estimate', () => { ) }) + it('charges a long reasoning row only for the one-line trigger it shows collapsed', () => { + const summary = 'x'.repeat(4_129) + const reasoning = estimateNativeChatRowHeight( + nativeChatRowContentMetrics(message(summary, 'reasoning')), + NO_CHROME + ) + expect(reasoning).toBe(24) + expect( + estimateNativeChatRowHeight(nativeChatRowContentMetrics(message(summary)), NO_CHROME) + ).toBeGreaterThan(900) + }) + it('reuses one derivation per message', () => { const subject = message('cached') expect(nativeChatRowContentMetrics(subject)).toBe(nativeChatRowContentMetrics(subject)) diff --git a/src/renderer/src/components/native-chat/native-chat-row-height-estimate.ts b/src/renderer/src/components/native-chat/native-chat-row-height-estimate.ts index ca60ed38e0e..9a5f052dff1 100644 --- a/src/renderer/src/components/native-chat/native-chat-row-height-estimate.ts +++ b/src/renderer/src/components/native-chat/native-chat-row-height-estimate.ts @@ -42,6 +42,8 @@ const PROSE_MIN_LINES = 1 const USER_BUBBLE_CHROME_PX = 32 const IMAGE_STRIP_PX = 88 const TOOL_RUN_PX = 40 +/** A reasoning row draws only its `xs` trigger button until opened; opening remeasures it. */ +const COLLAPSED_REASONING_PX = 24 const SUBAGENT_ROW_PX = 32 /** The one-line head that names a subagent above its own rows. */ export const NATIVE_CHAT_SUBAGENT_SECTION_HEAD_PX = SUBAGENT_ROW_PX @@ -122,7 +124,12 @@ export function estimateNativeChatRowHeight( height = content.subagentGroupCount * SUBAGENT_ROW_PX partCount = height > 0 ? 1 : 0 } else { - height = content.textLines * typography.lineHeightPx + height = + content.role === 'reasoning' + ? content.textLines > 0 + ? COLLAPSED_REASONING_PX + : 0 + : content.textLines * typography.lineHeightPx if (content.role === 'user' && content.textLines > 0) { height += USER_BUBBLE_CHROME_PX } diff --git a/src/renderer/src/components/native-chat/native-chat-subagent-section-slots.ts b/src/renderer/src/components/native-chat/native-chat-subagent-section-slots.ts index 0d5ca520677..55d5514ecc3 100644 --- a/src/renderer/src/components/native-chat/native-chat-subagent-section-slots.ts +++ b/src/renderer/src/components/native-chat/native-chat-subagent-section-slots.ts @@ -110,6 +110,7 @@ export function nativeChatSubagentSectionSlots({ receipt, status: undefined, folded: false, + drawsMessage: true, turnFolds: false, turnDiff: undefined, subagentRoster: undefined, diff --git a/src/renderer/src/components/native-chat/native-chat-subagent-sections.test.ts b/src/renderer/src/components/native-chat/native-chat-subagent-sections.test.ts index f483507c97e..323bd0d820f 100644 --- a/src/renderer/src/components/native-chat/native-chat-subagent-sections.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-subagent-sections.test.ts @@ -240,6 +240,17 @@ describe("a subagent's rows live in its own section", () => { expect(new Set(slots.map(nativeChatSlotKey)).size).toBe(slots.length) }) + // The parent's live line speaks for the session's own agent only, so a child's open block draws. + it("draws a working agent's open reasoning in its section", () => { + const thinking = row('child-think', say('Comparing the two diffs'), { + ...by('task-1'), + role: 'reasoning', + state: 'running' + }) + const rows = [...transcriptWith('working').slice(0, 3), thinking] + expect(outline(slotsOf(rows, { 'task-1': true }, true))).toContain('>child-think') + }) + // The fold reads only the conversation, so a subagent's failure after the answer is its own. it("folds a settled turn to the session's own answer, not a subagent's later failure", () => { const failed = [{ type: 'text' as const, text: 'The subagent failed.', tone: 'error' as const }] diff --git a/src/renderer/src/components/native-chat/native-chat-transcript-slots.test.ts b/src/renderer/src/components/native-chat/native-chat-transcript-slots.test.ts index 7841e2de190..1d7338f21ac 100644 --- a/src/renderer/src/components/native-chat/native-chat-transcript-slots.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-transcript-slots.test.ts @@ -542,4 +542,19 @@ describe('turn-owned grouping', () => { ['C', false] ]) }) + + it('skips only the open reasoning the live line discloses, and only while it does', () => { + const reasoning = (id: string, state: 'running' | 'completed'): NativeChatMessage => ({ + ...text(id, 'Weighing two approaches', 'reasoning'), + state + }) + const live = { turnKeys: ['A', 'A', 'A'], liveTurnKey: 'A', isWorking: true } + const rows = [text('A', 'go', 'user'), reasoning('r-1', 'running'), reasoning('r-2', 'running')] + const ids = (overrides: Partial[1]>) => + build(rows, { ...live, ...overrides }).map((slot) => slot.message.id) + expect(ids({ liveReasoningId: 'r-2' })).toEqual(['A', 'r-1']) + // Nothing discloses it (a prompt took the line, or it says something else): it draws. + expect(ids({ liveReasoningId: null })).toEqual(['A', 'r-1', 'r-2']) + expect(ids({})).toEqual(['A', 'r-1', 'r-2']) + }) }) diff --git a/src/renderer/src/components/native-chat/native-chat-transcript-slots.ts b/src/renderer/src/components/native-chat/native-chat-transcript-slots.ts index 1069b99d00d..e07682b15a5 100644 --- a/src/renderer/src/components/native-chat/native-chat-transcript-slots.ts +++ b/src/renderer/src/components/native-chat/native-chat-transcript-slots.ts @@ -14,7 +14,10 @@ import { type NativeChatMessage } from '../../../../shared/native-chat-types' import type { NativeChatTurnStatus } from '../../../../shared/native-chat-turn-status' -import { nativeChatMessagesWaitingBehindLiveTurn } from '../../../../shared/native-chat-turn-membership' +import { + isNativeChatRowInLiveWorkingTurn, + nativeChatMessagesWaitingBehindLiveTurn +} from '../../../../shared/native-chat-turn-membership' import { nativeChatTurnBarRows } from '../../../../shared/native-chat-turn-grouping' import { nativeChatTurnFold, @@ -68,6 +71,9 @@ export type NativeChatMessageSlot = { /** This row is behind its turn's folded status row: it draws no prose and no * tool activity, only work that outlives the turn. */ folded: boolean + /** Whether the message itself draws. False when the slot is here only for its turn's bar or diff + * rollup: a folded or empty row, or the open block the live line shows. */ + drawsMessage: boolean /** Whether this row's turn hides anything, so its status row offers a caret. */ turnFolds: boolean turnDiff: NativeChatTurnDiff | undefined @@ -100,6 +106,8 @@ export type NativeChatTranscriptSlotsInput = { lifecycleWorking: boolean subagentSections?: NativeChatSubagentSections subagentChoices?: NativeChatSubagentChoices + /** The open reasoning block the live activity line discloses: its row takes no slot meanwhile. */ + liveReasoningId?: string | null } export function buildNativeChatTranscriptSlots( @@ -117,7 +125,8 @@ export function buildNativeChatTranscriptSlots( isWorking, lifecycleWorking, subagentSections: sections = NO_NATIVE_CHAT_SUBAGENT_SECTIONS, - subagentChoices: choices = NO_NATIVE_CHAT_SUBAGENT_CHOICES + subagentChoices: choices = NO_NATIVE_CHAT_SUBAGENT_CHOICES, + liveReasoningId = null } = input // One pass to decide what each row draws, then the fold over those readings — // so "is this the answer" and "does this row render prose" cannot disagree. @@ -191,24 +200,27 @@ export function buildNativeChatTranscriptSlots( // Skipping a folded row entirely is what keeps windowing honest: a counted // index the row declines to draw reserves estimated height for nothing and // opens a gap in the transcript. + const activeTurnIsWorking = isNativeChatRowInLiveWorkingTurn( + turnKey, + liveTurnKey, + isWorking || lifecycleWorking + ) const drawsRow = - receipt !== undefined || (!folded && nativeChatRowRendersContent(message.blocks)) + receipt !== undefined || + (!folded && nativeChatRowRendersContent(message.blocks) && message.id !== liveReasoningId) const roster = sectionSlots.rosterAt(message.id) if (drawsRow || status !== undefined || turnDiff !== undefined) { slots.push({ kind: 'message', message, turnKey, - // Liveness is the owning turn's, not the newest prompt's: a running turn's - // rows stay live while a newer message waits behind it. - activeTurnIsWorking: - (liveTurnKey ? turnKey === liveTurnKey : turnKey === undefined) && - (isWorking || lifecycleWorking), + activeTurnIsWorking, trailingRun: index === trailingRunIndex, receipt, status: status ?? undefined, statusAbove: bar?.above === true && status !== undefined, folded, + drawsMessage: drawsRow, turnFolds: turnKey !== undefined && foldableTurnKeys.has(turnKey), turnDiff, subagentRoster: roster, diff --git a/src/renderer/src/components/native-chat/use-native-chat-transcript-slots.ts b/src/renderer/src/components/native-chat/use-native-chat-transcript-slots.ts new file mode 100644 index 00000000000..c6b5a26812f --- /dev/null +++ b/src/renderer/src/components/native-chat/use-native-chat-transcript-slots.ts @@ -0,0 +1,87 @@ +import { useMemo } from 'react' +import { + nativeChatLiveLine, + type NativeChatLiveLine +} from '../../../../shared/native-chat-live-line' +import { isNativeChatRowInLiveWorkingTurn } from '../../../../shared/native-chat-turn-membership' +import { + buildNativeChatTranscriptSlots, + type NativeChatTranscriptSlot, + type NativeChatTranscriptSlotsInput +} from './native-chat-transcript-slots' + +/** The transcript's slots and its live activity line, decided together: the open reasoning block + * the line discloses takes no slot, so a row is hidden exactly while the line shows it. */ +export function useNativeChatTranscriptSlots({ + line, + ...input +}: Omit & { + line: { draws: boolean; thinking: boolean; activityText?: string | null } +}): { slots: NativeChatTranscriptSlot[]; liveLine: NativeChatLiveLine | null } { + const { + messages, + typography, + turnKeys, + liveTurnKey, + receipts, + turnStatuses, + turnDiffs, + expandedTurnKeys, + isWorking, + lifecycleWorking, + subagentSections, + subagentChoices + } = input + const { draws, thinking, activityText } = line + const liveLine = useMemo( + () => + nativeChatLiveLine({ + draws, + thinking, + activityText, + messages, + inLiveWorkingTurn: (index) => + isNativeChatRowInLiveWorkingTurn( + turnKeys[index], + liveTurnKey, + isWorking || lifecycleWorking + ) + }), + [activityText, draws, isWorking, lifecycleWorking, liveTurnKey, messages, thinking, turnKeys] + ) + const liveReasoningId = liveLine?.reasoning?.message.id ?? null + const slots = useMemo( + () => + buildNativeChatTranscriptSlots({ + messages, + typography, + turnKeys, + liveTurnKey, + receipts, + turnStatuses, + turnDiffs, + expandedTurnKeys, + isWorking, + lifecycleWorking, + subagentSections, + subagentChoices, + liveReasoningId + }), + [ + expandedTurnKeys, + isWorking, + lifecycleWorking, + liveReasoningId, + liveTurnKey, + messages, + receipts, + subagentChoices, + subagentSections, + turnDiffs, + turnKeys, + turnStatuses, + typography + ] + ) + return { slots, liveLine } +} diff --git a/src/renderer/src/components/native-chat/use-native-chat-transcript-window.options.test.tsx b/src/renderer/src/components/native-chat/use-native-chat-transcript-window.options.test.tsx index 0a8f6347ce9..257e507c697 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-transcript-window.options.test.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-transcript-window.options.test.tsx @@ -57,6 +57,7 @@ function slot(id: string): NativeChatMessageSlot { receipt: undefined, status: undefined, folded: false, + drawsMessage: true, turnFolds: false, turnDiff: undefined, subagentRoster: undefined, diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 7ff593c472d..fe2aff1199c 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -17763,6 +17763,7 @@ "empty": "No users found" }, "native-chat": { + "reasoning": "Reasoning", "failureWords": { "theAgent": "The agent", "providerStartFailed": "{{agent}} stopped before it finished starting.", @@ -18320,6 +18321,8 @@ "queuePaused": "Queue paused", "resume": "Resume" }, + "thought": "Thought", + "thoughtForDuration": "Thought for {{duration}}", "structuredSessionHostDeclined": "Opened {{value0}} in a terminal", "structuredSessionHostDeclinedDescription": "This server can't run a {{value0}} chat in this workspace.", "structuredSessionHostUnreachable": "Could not reach {{value0}}", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 9708bdc6ba1..cd2f8913930 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -17533,6 +17533,7 @@ }, "components": { "native-chat": { + "reasoning": "Razonamiento", "contextMenu": { "copyOrcaSessionId": "Copiar ID de sesión de Orca", "orcaSessionIdCopied": "ID de sesión de Orca copiado", @@ -17951,6 +17952,8 @@ "queuePaused": "Cola en pausa", "resume": "Reanudar" }, + "thought": "Pensó", + "thoughtForDuration": "Pensó durante {{duration}}", "receipt": { "cancelled": "Cancelado", "unavailable": "Respuesta seleccionada no disponible", diff --git a/src/renderer/src/i18n/locales/fr.json b/src/renderer/src/i18n/locales/fr.json index 3bc1a96946e..0ca533b16d7 100644 --- a/src/renderer/src/i18n/locales/fr.json +++ b/src/renderer/src/i18n/locales/fr.json @@ -17632,6 +17632,7 @@ }, "components": { "native-chat": { + "reasoning": "Raisonnement", "contextMenu": { "copyOrcaSessionId": "Copier l'ID de session Orca", "orcaSessionIdCopied": "ID de session Orca copié", @@ -18168,6 +18169,8 @@ "queuePaused": "File d'attente en pause", "resume": "Reprendre" }, + "thought": "A réfléchi", + "thoughtForDuration": "A réfléchi pendant {{duration}}", "structuredSessionHostDeclined": "{{value0}} ouvert dans un terminal", "structuredSessionHostDeclinedDescription": "Ce serveur ne peut pas exécuter de discussion {{value0}} dans cet espace de travail.", "structuredSessionHostUnreachable": "Impossible de joindre {{value0}}", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index fda99065b37..7abc80e5c0b 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -17568,6 +17568,7 @@ } }, "native-chat": { + "reasoning": "推論", "contextMenu": { "copyOrcaSessionId": "Orca セッション ID をコピー", "orcaSessionIdCopied": "Orca セッション ID をコピーしました", @@ -18104,6 +18105,8 @@ "queuePaused": "キューは一時停止中です", "resume": "再開" }, + "thought": "考えました", + "thoughtForDuration": "{{duration}} 考えました", "structuredSessionHostDeclined": "{{value0}} をターミナルで開きました", "structuredSessionHostDeclinedDescription": "このサーバーはこのワークスペースで {{value0}} チャットを実行できません。", "structuredSessionHostUnreachable": "{{value0}} に接続できませんでした", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index e06aba01d15..73330f8e45f 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -17568,6 +17568,7 @@ } }, "native-chat": { + "reasoning": "추론", "contextMenu": { "copyOrcaSessionId": "Orca 세션 ID 복사", "orcaSessionIdCopied": "Orca 세션 ID를 복사했습니다", @@ -18104,6 +18105,8 @@ "queuePaused": "대기열이 일시 중지되었습니다", "resume": "재개" }, + "thought": "생각함", + "thoughtForDuration": "{{duration}} 동안 생각함", "structuredSessionHostDeclined": "{{value0}}을(를) 터미널에서 열었습니다", "structuredSessionHostDeclinedDescription": "이 서버는 이 워크스페이스에서 {{value0}} 채팅을 실행할 수 없습니다.", "structuredSessionHostUnreachable": "{{value0}}에 연결할 수 없습니다", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 36efa599608..d141cf70610 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -17533,6 +17533,7 @@ }, "components": { "native-chat": { + "reasoning": "推理", "contextMenu": { "copyOrcaSessionId": "复制 Orca 会话 ID", "orcaSessionIdCopied": "已复制 Orca 会话 ID", @@ -18069,6 +18070,8 @@ "queuePaused": "队列已暂停", "resume": "继续" }, + "thought": "已思考", + "thoughtForDuration": "思考了 {{duration}}", "structuredSessionHostDeclined": "已在终端中打开 {{value0}}", "structuredSessionHostDeclinedDescription": "此服务器无法在此工作区中运行 {{value0}} 聊天。", "structuredSessionHostUnreachable": "无法连接到 {{value0}}", diff --git a/src/shared/agent-session-journal-schemas.test.ts b/src/shared/agent-session-journal-schemas.test.ts index 10dac12611b..6ff5e56e30c 100644 --- a/src/shared/agent-session-journal-schemas.test.ts +++ b/src/shared/agent-session-journal-schemas.test.ts @@ -23,6 +23,18 @@ const RESOLUTION = { // Canonical fixtures are typed: if a shape here stops compiling, the schema // audit below is validating the wrong model. const CANONICAL_BODIES: AgentJournalItemBody[] = [ + { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Inspecting the request' }] + }, + { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Inspecting the request' }], + state: 'completed', + completedAt: 2_000 + }, { kind: 'message', role: 'user', @@ -475,6 +487,14 @@ describe('thread goal fields', () => { sentAs: 'scheduled' }) ).toBe(true) + expect( + isAdmissibleAgentJournalItemBody({ + kind: 'message', + role: 'reasoning', + blocks: [], + state: 'paused' + }) + ).toBe(true) expect( isAdmissibleAgentJournalItemBody({ kind: 'status', @@ -495,6 +515,8 @@ describe('thread goal fields', () => { for (const body of [ { kind: 'message', role: 'user', blocks: [], sentAs: 5 }, { kind: 'message', role: 'user', blocks: [], sentAs: '' }, + { kind: 'message', role: 'reasoning', blocks: [], state: '' }, + { kind: 'message', role: 'reasoning', blocks: [], completedAt: 'later' }, { kind: 'status', text: 'Goal set', threadGoal: { state: 'set' } }, { kind: 'status', diff --git a/src/shared/agent-session-journal-schemas.ts b/src/shared/agent-session-journal-schemas.ts index 89118307dd6..3cbb259af36 100644 --- a/src/shared/agent-session-journal-schemas.ts +++ b/src/shared/agent-session-journal-schemas.ts @@ -27,6 +27,7 @@ import { z } from 'zod' import { AgentSessionContextUsageSchema } from './agent-session-context-usage-schema' +import { AgentJournalThreadGoalStateSchema } from './agent-session-journal-thread-goal-schema' import { AgentSessionFailureFactSchema } from './agent-session-failure-fact-schema' import { knownTags, openDiscriminatedUnion } from './agent-session-journal-open-union' import type { @@ -180,27 +181,12 @@ const MessageBody = z.object({ blocks: z.array(Block), // Open like roles: a send mode a newer build writes must not turn the row malformed. sentAs: z.string().min(1).optional(), - command: z.object({ name: z.string().min(1) }).optional() + command: z.object({ name: z.string().min(1) }).optional(), + // Open like `sentAs`: a state a newer host writes reads as completed, never malformed. + state: z.string().min(1).optional(), + completedAt: z.number().finite().optional() }) -const ThreadGoal = z.object({ - objective: z.string(), - status: z.string().min(1), - tokenBudget: z.number().finite().nullable(), - tokensUsed: z.number().finite(), - timeUsedSeconds: z.number().finite(), - createdAt: z.number().finite(), - updatedAt: z.number().finite() -}) - -/** Like blocks: an unknown `state` stays admissible, a known one with a broken payload does not. */ -const ThreadGoalState = openDiscriminatedUnion( - z.discriminatedUnion('state', [ - z.object({ state: z.literal('set'), goal: ThreadGoal }), - z.object({ state: z.literal('cleared') }) - ]) -) - /** A turn's lifecycle, as the turn item and the legacy status row both carry it. */ const TurnLifecycleFields = { turnId: z.string(), @@ -259,7 +245,7 @@ const KnownItemBody = z.discriminatedUnion('kind', [ tone: z.string().optional(), turnLifecycle: z.object(TurnLifecycleFields).optional(), providerFrame: ProviderFrame.optional(), - threadGoal: ThreadGoalState.optional(), + threadGoal: AgentJournalThreadGoalStateSchema.optional(), failure: AgentSessionFailureFactSchema.optional() }), z.object({ diff --git a/src/shared/agent-session-journal-thread-goal-schema.ts b/src/shared/agent-session-journal-thread-goal-schema.ts new file mode 100644 index 00000000000..af86675fa0d --- /dev/null +++ b/src/shared/agent-session-journal-thread-goal-schema.ts @@ -0,0 +1,22 @@ +// The journal's thread-goal transition, validated as deeply as the rest of the render model. + +import { z } from 'zod' +import { openDiscriminatedUnion } from './agent-session-journal-open-union' + +const ThreadGoal = z.object({ + objective: z.string(), + status: z.string().min(1), + tokenBudget: z.number().finite().nullable(), + tokensUsed: z.number().finite(), + timeUsedSeconds: z.number().finite(), + createdAt: z.number().finite(), + updatedAt: z.number().finite() +}) + +/** Like blocks: an unknown `state` stays admissible, a known one with a broken payload does not. */ +export const AgentJournalThreadGoalStateSchema = openDiscriminatedUnion( + z.discriminatedUnion('state', [ + z.object({ state: z.literal('set'), goal: ThreadGoal }), + z.object({ state: z.literal('cleared') }) + ]) +) diff --git a/src/shared/agent-session-journal-types.ts b/src/shared/agent-session-journal-types.ts index ff5d53aca76..a827d65296a 100644 --- a/src/shared/agent-session-journal-types.ts +++ b/src/shared/agent-session-journal-types.ts @@ -100,6 +100,11 @@ export type AgentJournalBoundedPayload = { export const AGENT_JOURNAL_MESSAGE_SEND_MODES = ['goal'] as const export type AgentJournalMessageSendMode = (typeof AGENT_JOURNAL_MESSAGE_SEND_MODES)[number] +/** Whether the provider is still producing a message. Persisted and open for growth: a reader + * that cannot place a value reads it as `completed`. */ +export const AGENT_JOURNAL_MESSAGE_STATES = ['running', 'completed'] as const +export type AgentJournalMessageState = (typeof AGENT_JOURNAL_MESSAGE_STATES)[number] + export type AgentJournalMessageItem = { kind: 'message' role: NativeChatRole @@ -110,6 +115,13 @@ export type AgentJournalMessageItem = { /** Present on a conversation command the user sent, such as `/compact`. The text is what the * user typed; this names the command so no reader parses it. Open like `sentAs`. */ command?: { name: string } + /** Written on reasoning rows. ABSENT MEANS UNKNOWN — an older host, or a row from before the + * field — and never reads as live. The row's `observedAt` is when it started. */ + state?: AgentJournalMessageState + /** Host clock when the host saw the message end: its own end, or the end of the turn or + * stream that cut it off. Absent only when no end was seen live — history, a crash sweep — so + * no duration is claimed. */ + completedAt?: number } export type AgentJournalToolCallState = 'running' | 'completed' | 'failed' diff --git a/src/shared/native-chat-live-line.test.ts b/src/shared/native-chat-live-line.test.ts new file mode 100644 index 00000000000..981b9db2f23 --- /dev/null +++ b/src/shared/native-chat-live-line.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from 'vitest' +import { nativeChatLiveLine } from './native-chat-live-line' +import type { NativeChatMessage } from './native-chat-types' + +const prompt: NativeChatMessage = { + id: 'user-1', + role: 'user', + blocks: [{ type: 'text', text: 'Start the task' }], + timestamp: 1_000, + source: 'transcript' +} +const open: NativeChatMessage = { + id: 'r-1', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }], + timestamp: 1_000, + source: 'transcript', + state: 'running' +} +const line = (fields: { draws?: boolean; thinking?: boolean; activityText?: string | null }) => + nativeChatLiveLine({ + draws: true, + thinking: true, + messages: [prompt, open], + inLiveWorkingTurn: () => true, + ...fields + }) + +describe('the live line both clients draw', () => { + it('discloses the open block while it reads "Thinking"', () => { + expect(line({})).toEqual({ + thinking: true, + activityText: null, + reasoning: { message: open, markdown: 'Weighing two approaches' } + }) + }) + + it('discloses nothing, so nothing is hidden, unless it draws and the turn is reasoning', () => { + // A prompt the reader owes replaces the line: no line, and the row draws. + expect(line({ draws: false })).toBeNull() + expect(line({ thinking: false })).toEqual({ + thinking: false, + activityText: null, + reasoning: null + }) + }) + + // The label may read the provider's own words; the block it discloses is still the open one. + it('keeps the activity text it was handed', () => { + expect(line({ activityText: 'Summarizing the plan' })).toMatchObject({ + activityText: 'Summarizing the plan', + reasoning: { message: open } + }) + }) +}) diff --git a/src/shared/native-chat-live-line.ts b/src/shared/native-chat-live-line.ts new file mode 100644 index 00000000000..d7c10d19f20 --- /dev/null +++ b/src/shared/native-chat-live-line.ts @@ -0,0 +1,38 @@ +// The live turn's tail line, as desktop and mobile both draw it: whether it draws, what it says, +// and which open reasoning block it discloses. One value, so a block's row is hidden exactly while +// the line that shows it draws. + +import { + selectNativeChatLiveReasoning, + type NativeChatLiveReasoning +} from './native-chat-reasoning-row' +import type { NativeChatMessage } from './native-chat-types' + +export type NativeChatLiveLine = { + /** The turn is reasoning now; the label reads "Thinking" unless activity text outranks it. */ + thinking: boolean + activityText: string | null + /** The open block the line discloses; its row draws nothing meanwhile. */ + reasoning: NativeChatLiveReasoning | null +} + +export function nativeChatLiveLine(input: { + /** The running turn's tail is the activity line: nothing (a prompt the reader owes) replaces it. */ + draws: boolean + thinking: boolean + activityText?: string | null + messages: readonly NativeChatMessage[] + inLiveWorkingTurn: (index: number) => boolean +}): NativeChatLiveLine | null { + if (!input.draws) { + return null + } + return { + thinking: input.thinking, + activityText: input.activityText ?? null, + // Rows never say "Thinking", so the line discloses only while it is the one that does. + reasoning: input.thinking + ? selectNativeChatLiveReasoning(input.messages, input.inLiveWorkingTurn) + : null + } +} diff --git a/src/shared/native-chat-reasoning-row.test.ts b/src/shared/native-chat-reasoning-row.test.ts new file mode 100644 index 00000000000..76c2f422e15 --- /dev/null +++ b/src/shared/native-chat-reasoning-row.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from 'vitest' +import { + nativeChatReasoningDisclosureKey, + nativeChatReasoningHeadline, + nativeChatReasoningHeadlineText, + selectNativeChatLiveReasoning +} from './native-chat-reasoning-row' +import type { NativeChatMessage } from './native-chat-types' + +function row( + id: string, + role: NativeChatMessage['role'], + text: string, + fields: Partial = {} +): NativeChatMessage { + return { + id, + role, + blocks: [{ type: 'text', text }], + timestamp: 1_000, + source: 'transcript', + ...fields + } +} + +const prompt = row('user-1', 'user', 'Start the task') +const open = row('r-1', 'reasoning', 'Weighing two approaches', { state: 'running' }) +const inTurn = (): boolean => true + +describe('the open block the live line discloses', () => { + const select = (messages: NativeChatMessage[], live: (index: number) => boolean = inTurn) => + selectNativeChatLiveReasoning(messages, live)?.message ?? null + + it('is the newest root reasoning row with text, with the text it has so far', () => { + expect(selectNativeChatLiveReasoning([prompt, open], inTurn)).toEqual({ + message: open, + markdown: 'Weighing two approaches' + }) + // A host that keeps no lifecycle gets the same single live slot. + const stateless = row('r-1', 'reasoning', 'Weighing two approaches') + expect(select([prompt, stateless])).toBe(stateless) + }) + + it('is nothing when the block is blank or ended', () => { + expect(select([prompt, row('r-1', 'reasoning', ' \n', { state: 'running' })])).toBeNull() + expect(select([prompt, { ...open, state: 'completed' }])).toBeNull() + }) + + it('is nothing once a tool or the answer is newer than the block', () => { + expect(select([prompt, open, row('a-1', 'assistant', 'Here it is')])).toBeNull() + const tool = row('t-1', 'assistant', '', { + blocks: [{ type: 'tool-call', name: 'Read', input: { file_path: 'a.ts' }, state: 'running' }] + }) + expect(select([prompt, open, tool])).toBeNull() + }) + + it('looks past notices, empty rows and rows outside the live working turn', () => { + const notice = row('s-1', 'system', 'Compacting') + const empty = row('a-0', 'assistant', '') + const waiting = row('user-2', 'user', 'Also say banana') + expect(select([prompt, open, notice, empty, waiting], (index) => index < 4)).toBe(open) + }) + + it('stops at the live turn prompt and ignores a subagent reasoning', () => { + expect(select([open, prompt])).toBeNull() + const child = row('r-2', 'reasoning', 'Child thinking', { state: 'running', agentId: 'sub-1' }) + expect(select([prompt, child])).toBeNull() + expect(select([prompt, open, child])).toBe(open) + }) + + it('keys one block the same for the line and the row, apart from tool runs', () => { + expect(nativeChatReasoningDisclosureKey('r-1')).toBe('reasoning:r-1') + }) +}) + +describe('the reasoning headline', () => { + const text = ( + fields: { state?: 'running' | 'completed'; completedAt?: number }, + live = false + ): string => + nativeChatReasoningHeadlineText( + nativeChatReasoningHeadline({ timestamp: 1_000, ...fields }, { live }) + ) + + it('reads the span the host saw, at least a second, and claims none it did not see', () => { + expect(text({ state: 'completed', completedAt: 66_000 })).toBe('Thought for 1m 5s') + expect(text({ state: 'completed', completedAt: 1_300 })).toBe('Thought for 1s') + expect(text({ state: 'completed' })).toBe('Thought') + expect(text({ state: 'running' })).toBe('Thought') + expect(text({})).toBe('Reasoning') + }) + + it('claims no past tense for a block not yet ended inside its working turn', () => { + expect(text({ state: 'running' }, true)).toBe('Reasoning') + expect(text({}, true)).toBe('Reasoning') + expect(text({ state: 'completed', completedAt: 13_000 }, true)).toBe('Thought for 12s') + }) +}) diff --git a/src/shared/native-chat-reasoning-row.ts b/src/shared/native-chat-reasoning-row.ts new file mode 100644 index 00000000000..8bcd14b6ea1 --- /dev/null +++ b/src/shared/native-chat-reasoning-row.ts @@ -0,0 +1,98 @@ +// The reasoning row, as desktop and mobile both draw it: which open block the live activity line +// discloses, and what a row's collapsed headline says. Read from host facts only — the row's start +// (`timestamp`) and the end the host saw — so every client tells the same story about one row. + +import { isRootAgentJournalItem } from './agent-session-journal-producer' +import { deriveNativeChatRowContent, nativeChatRowRendersContent } from './native-chat-row-content' +import { formatNativeChatDuration } from './native-chat-turn-status' +import type { NativeChatMessage } from './native-chat-types' + +/** An open reasoning block and the text it has so far. */ +export type NativeChatLiveReasoning = { message: NativeChatMessage; markdown: string } + +/** + * The agent's open reasoning block, with its text, when it is the newest thing its live working + * turn produced; else null. Only `nativeChatLiveLine` asks, so a block is disclosed (and its row + * hidden) only while the line that discloses it draws. + */ +export function selectNativeChatLiveReasoning( + messages: readonly NativeChatMessage[], + inLiveWorkingTurn: (index: number) => boolean +): NativeChatLiveReasoning | null { + for (let index = messages.length - 1; index >= 0; index -= 1) { + const message = messages[index] + if (!message || !inLiveWorkingTurn(index)) { + continue + } + if (message.role === 'user') { + return null + } + // A notice is not newer content, and a subagent's reasoning draws in its own section. + if (message.role === 'system' || !isRootAgentJournalItem(message)) { + continue + } + if (message.role !== 'reasoning') { + if (!nativeChatRowRendersContent(message.blocks)) { + continue + } + return null + } + // Not `=== 'running'`: a host that keeps no lifecycle gets the same single live slot. + if (message.state === 'completed') { + return null + } + const { markdown } = deriveNativeChatRowContent(message.blocks) + return markdown.trim().length > 0 ? { message, markdown } : null + } + return null +} + +/** One key per reasoning block, read by the live line and the finished row alike. */ +export function nativeChatReasoningDisclosureKey(messageId: string): string { + return `reasoning:${messageId}` +} + +export type NativeChatReasoningHeadline = + /** From a host that kept no lifecycle, or not yet ended: nothing is claimed. */ + | { kind: 'reasoning' } + /** Ended, with no span the host saw. */ + | { kind: 'thought' } + | { kind: 'thoughtFor'; duration: string } + +/** `live`: the row is drawn inside its working turn or working subagent. */ +export function nativeChatReasoningHeadline( + message: Pick, + { live }: { live: boolean } +): NativeChatReasoningHeadline { + // Not ended yet, so it claims no past tense. + if (message.state === undefined || (live && message.state !== 'completed')) { + return { kind: 'reasoning' } + } + // An open row in a turn that is no longer live ended unseen, so it claims no duration. + if ( + message.state !== 'completed' || + message.completedAt === undefined || + message.timestamp === null + ) { + return { kind: 'thought' } + } + return { + kind: 'thoughtFor', + duration: formatNativeChatDuration( + Math.max(1, (message.completedAt - message.timestamp) / 1000) + ) + } +} + +/** English copy for clients without a translation catalog; desktop translates the same three. */ +const NATIVE_CHAT_REASONING_COPY = { + reasoning: 'Reasoning', + thought: 'Thought', + thoughtFor: (duration: string) => `Thought for ${duration}` +} as const + +export function nativeChatReasoningHeadlineText(headline: NativeChatReasoningHeadline): string { + return headline.kind === 'thoughtFor' + ? NATIVE_CHAT_REASONING_COPY.thoughtFor(headline.duration) + : NATIVE_CHAT_REASONING_COPY[headline.kind] +} diff --git a/src/shared/native-chat-tool-icon.ts b/src/shared/native-chat-tool-icon.ts index 9da8672374a..0a34703e36e 100644 --- a/src/shared/native-chat-tool-icon.ts +++ b/src/shared/native-chat-tool-icon.ts @@ -44,6 +44,8 @@ export type NativeChatToolIconName = * it names no tool category, because that row stands for a question rather * than for the call that asked it. */ | 'message-square-more' + /** The reasoning row's glyph, carried for the same aligned slot; it names no tool category. */ + | 'brain' /** Category to glyph. */ export const NATIVE_CHAT_TOOL_ICON_NAMES: Record = { diff --git a/src/shared/native-chat-turn-membership.ts b/src/shared/native-chat-turn-membership.ts index f00923996c4..b40290ae600 100644 --- a/src/shared/native-chat-turn-membership.ts +++ b/src/shared/native-chat-turn-membership.ts @@ -237,6 +237,16 @@ export function nativeChatMessagesWaitingBehindLiveTurn( ) } +/** Whether a row belongs to the turn running now. Liveness is the owning turn's, not the newest + * prompt's: a running turn's rows stay live while a newer message waits behind it. */ +export function isNativeChatRowInLiveWorkingTurn( + turnKey: string | undefined, + liveTurnKey: string | undefined, + working: boolean +): boolean { + return working && (liveTurnKey ? turnKey === liveTurnKey : turnKey === undefined) +} + function commandTurnRunning(items: readonly AgentJournalRenderItem[]): boolean { const running = liveStructuredAgentSessionTurnScope(items) const bodyOf = (itemId: string) => items.find((item) => item.itemId === itemId)?.body diff --git a/src/shared/native-chat-types.ts b/src/shared/native-chat-types.ts index e8315c7c873..46c82b7fabf 100644 --- a/src/shared/native-chat-types.ts +++ b/src/shared/native-chat-types.ts @@ -14,6 +14,7 @@ import type { AgentSessionTokenUsage } from './agent-session-context-usage' import type { AgentSessionFailureFact } from './agent-session-failure' import type { AgentJournalMessageSendMode, + AgentJournalMessageState, AgentJournalPosition, AgentJournalProducerLinkage, AgentJournalToolCallEnding, @@ -226,6 +227,10 @@ export type NativeChatMessage = AgentJournalProducerLinkage & { parentId?: string /** How a user message was delivered when it was not an ordinary prompt. */ sentAs?: AgentJournalMessageSendMode + /** The journal row's own lifecycle; absent means unknown, never live. */ + state?: AgentJournalMessageState + /** Host clock when the row's message was seen to end; absent when no end was seen live. */ + completedAt?: number /** On a conversation command the user sent, such as `/compact`: the command it names. */ command?: { name: string } /** Accepted but not yet handed to the agent: drawn after everything the agent has done. */ diff --git a/src/shared/structured-agent-session-live-turn.test.ts b/src/shared/structured-agent-session-live-turn.test.ts index 42f8ade61ba..ba48ab3b481 100644 --- a/src/shared/structured-agent-session-live-turn.test.ts +++ b/src/shared/structured-agent-session-live-turn.test.ts @@ -30,6 +30,20 @@ describe('isStructuredAgentSessionThinking', () => { expect(isStructuredAgentSessionThinking([turnStart, reasoning(2)])).toBe(true) }) + it("reads a reasoning row's own state when its host keeps one", () => { + const withState = (state: 'running' | 'completed'): AgentJournalRenderItem => + item('reasoning-state', 2, { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }], + state, + ...(state === 'completed' ? { completedAt: 3 } : {}) + }) + expect(isStructuredAgentSessionThinking([turnStart, withState('running')])).toBe(true) + // Ended reasoning stays the newest row while Claude streams a tool's input after it. + expect(isStructuredAgentSessionThinking([turnStart, withState('completed')])).toBe(false) + }) + it('is false once a tool call, a message or a diff lands after the reasoning', () => { const after = (body: AgentJournalRenderItem['body']): boolean => isStructuredAgentSessionThinking([turnStart, reasoning(2), item('after', 3, body)]) diff --git a/src/shared/structured-agent-session-live-turn.ts b/src/shared/structured-agent-session-live-turn.ts index 56db1518214..f630d39ebe6 100644 --- a/src/shared/structured-agent-session-live-turn.ts +++ b/src/shared/structured-agent-session-live-turn.ts @@ -133,7 +133,9 @@ export function isStructuredAgentSessionThinking( continue } if (body?.kind === 'message') { - newestContentIsReasoning = body.role === 'reasoning' + // A row that says it ended is not reasoning now; a host that keeps no state says nothing. + newestContentIsReasoning = + body.role === 'reasoning' && (body.state === undefined || body.state === 'running') } else if ( body?.kind === 'tool-call' || body?.kind === 'diff' || diff --git a/src/shared/structured-agent-session-projection.ts b/src/shared/structured-agent-session-projection.ts index 0fbb057f66e..6eee55bbeda 100644 --- a/src/shared/structured-agent-session-projection.ts +++ b/src/shared/structured-agent-session-projection.ts @@ -5,6 +5,8 @@ import { } from './agent-status-field-normalization' import { AGENT_JOURNAL_MESSAGE_SEND_MODES, + AGENT_JOURNAL_MESSAGE_STATES, + type AgentJournalMessageItem, type AgentJournalMessageSendMode, type AgentJournalRenderItem, type AgentJournalSubmission @@ -136,6 +138,20 @@ function isAgentJournalMessageSendMode(value: string): value is AgentJournalMess return AGENT_JOURNAL_MESSAGE_SEND_MODES.some((mode) => mode === value) } +/** A state this build cannot name reads as completed: a newer host's row is never live here. */ +function messageLifecycle( + body: AgentJournalMessageItem +): Pick { + const state: string | undefined = body.state + if (state === undefined) { + return {} + } + return { + state: AGENT_JOURNAL_MESSAGE_STATES.find((known) => known === state) ?? 'completed', + ...(body.completedAt !== undefined ? { completedAt: body.completedAt } : {}) + } +} + const projectedItems = new WeakMap() /** Deliberately NOT scoped by producer: every agent's rows are projected, and @@ -165,6 +181,7 @@ export function projectStructuredItemToNativeChat( // Reducer updates replace journal items, so unchanged rows keep their render caches. const projected = itemBlocks(item) const sentAs = item.body.kind === 'message' ? item.body.sentAs : undefined + const lifecycle = item.body.kind === 'message' ? messageLifecycle(item.body) : {} const command = item.body.kind === 'message' ? item.body.command : undefined const message: NativeChatMessage | null = projected ? { @@ -174,6 +191,7 @@ export function projectStructuredItemToNativeChat( blocks: projected.blocks, // A send mode this build cannot name renders as an ordinary message. ...(sentAs !== undefined && isAgentJournalMessageSendMode(sentAs) ? { sentAs } : {}), + ...lifecycle, ...(command ? { command } : {}) } : null diff --git a/src/shared/structured-agent-session-reasoning-projection.test.ts b/src/shared/structured-agent-session-reasoning-projection.test.ts new file mode 100644 index 00000000000..594c29504da --- /dev/null +++ b/src/shared/structured-agent-session-reasoning-projection.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from 'vitest' +import type { + AgentJournalMessageItem, + AgentJournalMessageState, + AgentJournalRenderItem +} from './agent-session-journal-types' +import { projectStructuredItemToNativeChat } from './structured-agent-session-projection' + +function item( + itemId: string, + sequence: number, + body: AgentJournalRenderItem['body'] +): AgentJournalRenderItem { + return { itemId, sequence, revision: 1, observedAt: sequence, body } +} + +describe('structured reasoning projection', () => { + it('preserves reasoning text and identity for desktop and mobile consumers', () => { + const blocks = [{ type: 'text' as const, text: 'Inspecting the request' }] + expect( + projectStructuredItemToNativeChat( + item('reasoning-1', 2, { + kind: 'message', + role: 'reasoning', + blocks + }) + ) + ).toEqual({ + id: 'reasoning-1', + role: 'reasoning', + blocks, + timestamp: 2, + journalPosition: { sequence: 2, index: 0 }, + source: 'transcript' + }) + }) + + it('carries a reasoning row lifecycle, and reads a state it cannot name as completed', () => { + const blocks = [{ type: 'text' as const, text: 'Inspecting' }] + const project = (body: Pick) => + projectStructuredItemToNativeChat( + item('reasoning-2', 3, { kind: 'message', role: 'reasoning', blocks, ...body }) + ) + expect(project({ state: 'running' })).toMatchObject({ state: 'running' }) + expect(project({ state: 'completed', completedAt: 9 })).toMatchObject({ + state: 'completed', + completedAt: 9 + }) + // oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: a newer host's state, which this build's type cannot name. + expect(project({ state: 'paused' as AgentJournalMessageState })).toMatchObject({ + state: 'completed' + }) + expect(project({})).not.toHaveProperty('state') + }) +})