Merge main and preserve mobile push reliability coverage

This commit is contained in:
Jinwoo-H
2026-09-09 03:24:06 -04:00
833 changed files with 35436 additions and 7229 deletions
+2
View File
@@ -227,6 +227,7 @@ jobs:
mapfile -t TEST_FILES < <(jq -r '.[] | select(
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
. != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts" and
. != "tests/e2e/local-ssh-browser-routing.spec.ts" and
. != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and
. != "tests/e2e/ssh-localhost.spec.ts" and
@@ -272,6 +273,7 @@ jobs:
if: >-
inputs.test_files == '' ||
inputs.ssh_source_changed == 'true' ||
contains(inputs.test_files, 'tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/local-ssh-browser-routing.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/ssh-startup-exec-readiness.spec.ts') ||
+341 -19
View File
@@ -17,15 +17,33 @@
"protection": "partial",
"owner": "runtime",
"layer": "service-integration",
"surfaces": ["headless startup", "push registration", "mobile notification replay"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "remote-runtime"],
"surfaces": [
"headless startup",
"push registration",
"native push delivery policy"
],
"platforms": [
"macos",
"linux",
"windows"
],
"providers": [
"local",
"remote-runtime"
],
"coveredPlatforms": [
"macos"
],
"coveredProviders": [
"local",
"remote-runtime"
],
"coverageNotes": "Actual startOrcad entry with mocked daemon/RPC startup boundaries; real controller, push service, and persisted device registry. Gateway send is stubbed.",
"motivatingLinks": ["https://github.com/stablyai/orca/pull/19204"],
"invariant": "Headless startup installs and disposes push delivery; push and replay preserve the three-minute away policy, and host activity cannot extend the seven-day mobile lease.",
"oracle": "Require registration after RPC identity initialization and shutdown cleanup; idle 179/180/0 yields false/true/false for socket and replay, and exactly one push. Persisted lease expires exactly at seven days and only explicit registration renews it.",
"motivatingLinks": [
"https://github.com/stablyai/orca/pull/19204"
],
"invariant": "Headless startup installs and disposes push delivery; native push is the only mobile OS-banner path, desktop notification categories remain authoritative, the three-minute away policy is preserved, and host activity cannot extend the seven-day mobile lease. The gateway emits individual notifications rather than custom summaries.",
"oracle": "Require registration after RPC identity initialization and shutdown cleanup; idle 179/180/0 yields false/true/false in retained event metadata and exactly one gateway push. Socket and reconnect paths retain app state and tray reconciliation without creating OS banners, desktop categories remain authoritative, and gateway alerts retain individual identities. Persisted lease expires exactly at seven days and only explicit registration renews it.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/mobile-notification-dismissal-store.test.ts src/renderer/src/hooks/useAutoAckViewedAgent.away.test.ts",
"ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/orcad/orcad-push-startup.test.ts src/main/runtime/push/push-policy-pipeline.integration.test.ts"
@@ -46,7 +64,7 @@
{
"file": "src/main/runtime/push/push-policy-pipeline.integration.test.ts",
"assertions": [
"carries the native idle boundary through replay, socket policy and push dispatch",
"carries the native idle boundary through replay and push dispatch",
"expires persisted registration at seven days despite host activity and renews explicitly"
]
}
@@ -78,15 +96,172 @@
"required": false,
"evidence": "Lifecycle and policy coverage; asserts exact gateway send counts and zero remaining dispatch listeners after shutdown."
},
"promotionCriteria": ["Collect repeated CI runs without unexplained failures."],
"promotionCriteria": [
"Collect repeated CI runs without unexplained failures."
],
"knownGaps": [
"Does not prove APNs silent background wakeup or actual operating-system idle transitions.",
"Rendererless agent/bell event generation remains outside the documented feature contract.",
"No live Windows or Linux policy evidence.",
"Native iOS dismissal callback processing is verified separately; real APNs background wakeup is not established. Coalesced summaries require membership-aware dismissal before they can be cleared safely."
"Native iOS dismissal callback processing is verified separately; real APNs background wakeup and Android automatic grouping are not established. Legacy summaries already delivered during rolling overlap rely on the bounded compatibility reader outside this gate."
],
"demotionRule": "Keep experimental if lifecycle or policy assertions fail; do not weaken them to bypass platform delivery gaps."
},
{
"id": "agent-session.history-forward-read-budget",
"title": "Journal catch-up reads only the next page and one lookahead row",
"maturity": "experimental",
"protection": "partial",
"owner": "agent-session-runtime",
"layer": "runtime-unit",
"surfaces": ["structured agent history", "structured agent subscriptions"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "ssh", "remote-runtime"],
"coverageNotes": "The real SQLite journal and production subscriber delivery are exercised with a folder workspace and remote host identity. The SQL and pagination code is shared across execution hosts; live SSH transport and Linux/Windows runtime execution are not exercised. PTY, daemon, WSL execution, and mobile rendering are unaffected.",
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts"
],
"invariant": "Forward catch-up preserves every item, revision, tombstone, sequence cursor, page byte bound, and reset behavior while reading at most the requested row count plus one from SQLite for each page.",
"oracle": "Reconnect a real subscriber to a 2,000-row journal and receive all 2,000 item identities in order through the live cursor; count the actual SQL rows returned and parsed as 2,009 instead of 11,000. Assert exact final-page hasNewer, unlimited reader compatibility, gap detection at the next page, and parse-stop behavior at the lookahead row. Existing history tests cover revisions, tombstones, byte-bound shrinking, epochs, and schema resets.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts src/main/native-chat/agent-session-journal"
],
"testFiles": [
"src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts",
"src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts",
"src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts"
],
"assertionRefs": [
{
"file": "src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts",
"assertions": [
"reconnects through every page with one lookahead row per page",
"keeps an exact final page final and preserves unlimited journal readers",
"reports a sequence gap when the next page reaches it",
"preserves parse-stop behavior at lookahead: %s"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts src/main/native-chat/agent-session-journal",
"result": "passed",
"durationSeconds": 9.94,
"summary": "214 tests passed across 19 files, including actual SQLite row and JSON parse counts through production subscriber catch-up."
}
],
"runtimeBudget": {
"p95Seconds": 30,
"scope": "Real SQLite journal unit and production subscriber tests; no launched app."
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial deterministic local validation; CI soak has not started."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "Before the change, SQL returned 2,000, 1,800, 1,600 through 200 rows across ten pages, failing the count assertion. The bounded query returns nine pages of 201 rows and a final 200, with exactly 2,009 row parses and identical item delivery."
},
"performanceBudget": {
"required": true,
"evidence": "Catch-up materialization and JSON parsing are linear in unseen journal rows plus page lookaheads. A cached parameterized LIMIT adds no polling, cache invalidation, output loss, protocol change, or provider calls."
},
"knownGaps": [
"Linux and Windows execution and live SSH transport have not been exercised.",
"The existing full reduced-state snapshot and batch projection cost are outside this SQL read budget."
],
"promotionCriteria": [
"Complete CI soak requirements while preserving the deterministic row budget and pagination oracles."
],
"demotionRule": "Keep experimental until CI soak; investigate fidelity or count failures without relaxing the row budget."
},
{
"id": "terminal-performance.osc-status-scan-budget",
"title": "OSC 9999 status bursts reuse forward terminator searches",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-runtime",
"layer": "shared-unit-and-runtime-unit",
"surfaces": ["terminal output ingestion", "terminal agent-status side effects"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "daemon", "ssh", "remote-runtime"],
"coverageNotes": "Shared parser tests cover provider-independent bytes; main and renderer contract tests cover status and terminal-output delivery. Live Linux, Windows, WSL, SSH and remote-runtime processes are not launched. Execution, liveness, paths, wire formats and mobile UI are unchanged.",
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/src/shared/agent-status-osc.ts"
],
"invariant": "Terminal status parsing preserves ordinary UTF-16 output, every valid payload in order, the last valid payload's clean-output offset, earliest BEL/ST termination, and incomplete-frame caps while searching each complete burst only forward.",
"oracle": "Two 5,000-frame bursts using exclusively BEL or ST produce every expected payload and ordinary output byte with at most twice the input length in native search ranges. Mixed terminators, every split through prefixes/JSON/ST, independent parser interleaving, malformed payloads, exact pending-cap boundaries and oversized complete frames retain their previous behavior. A one-character echo performs no terminator search. Parsed output chunks are not retained in legacy regular-expression state.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-scan-budget.test.ts src/shared/agent-status-types.test.ts src/main/runtime/orca-runtime-hook-agent-status-projection.test.ts src/renderer/src/components/terminal-pane/terminal-title-tracker-parity.test.ts src/renderer/src/components/terminal-pane/pty-connection-main-side-effect-authority.test.ts src/renderer/src/components/terminal-pane/pty-connection-hook-completion-side-effects.test.ts src/renderer/src/components/terminal-pane/pty-transport-eager-buffer-replay.test.ts"
],
"testFiles": [
"src/shared/agent-status-osc.test.ts",
"src/shared/agent-status-osc-scan-budget.test.ts",
"src/main/runtime/orca-runtime-hook-agent-status-projection.test.ts",
"src/renderer/src/components/terminal-pane/terminal-title-tracker-parity.test.ts"
],
"assertionRefs": [
{
"file": "src/shared/agent-status-osc-scan-budget.test.ts",
"assertions": [
"reads each burst only forward with terminator %j",
"keeps a one-character input echo on the ordinary-output path",
"does not retain the output chunk in legacy regular-expression state"
]
},
{
"file": "src/shared/agent-status-osc.test.ts",
"assertions": [
"uses the earliest mixed terminator and counts only parsed payload offsets",
"keeps a distant ST usable after many intervening BEL frames",
"preserves every split of prefixes, JSON, and both terminators across independent streams",
"applies the pending cap only to incomplete frames"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-scan-budget.test.ts src/shared/agent-status-types.test.ts src/main/runtime/orca-runtime-hook-agent-status-projection.test.ts src/renderer/src/components/terminal-pane/terminal-title-tracker-parity.test.ts src/renderer/src/components/terminal-pane/pty-connection-main-side-effect-authority.test.ts src/renderer/src/components/terminal-pane/pty-connection-hook-completion-side-effects.test.ts src/renderer/src/components/terminal-pane/pty-transport-eager-buffer-replay.test.ts",
"result": "passed",
"durationSeconds": 23.96,
"summary": "153 tests passed across eight files. Independent baseline differential review also matched 3,704 streams and 45,141 chunk results."
}
],
"runtimeBudget": {
"p95Seconds": 30,
"scope": "Shared parser and main/renderer terminal contract tests; no launched app."
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial deterministic local validation; CI soak has not started."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "The unchanged parser failed both search budgets: 618,560,785 searched characters for the 246,390-character BEL burst and 631,068,285 for the 251,390-character ST burst. Reusing forward match positions reduces those totals to 492,770 and 502,770 characters respectively, within twice the input length, with identical complete results."
},
"performanceBudget": {
"required": true,
"evidence": "Warmed Node 24 macOS CPU medians: a 250 KB / 5,000-status burst fell from 100.240 ms to 1.903 ms; a 1 MB / 20,000-status burst fell from 1,588.492 ms to 7.369 ms. Wall-clock medians were 134.878 to 2.484 ms and 2,536.927 to 12.902 ms under concurrent machine load. These are adverse bursts, not typical callback sizes. The ordinary-output path is unchanged; one-character echo CPU was 2.173 versus 2.342 ms per 100,000 calls, and single-status BEL CPU was 10.835 versus 10.897 ms per 30,000 calls. Both native terminator searches advance monotonically within the current chunk; no regex state retains the input. No scheduling, polling, provider calls, pending limits, output filtering or payload parsing changed."
},
"knownGaps": [
"Live Electron input latency and Linux/Windows/WSL/SSH execution have not been measured for this parser-only change.",
"Fragmented unterminated payload accumulation and downstream processing of large status arrays remain outside this complete-burst search budget."
],
"promotionCriteria": [
"Complete CI soak while preserving byte fidelity and deterministic search budgets."
],
"demotionRule": "Keep experimental until CI soak; investigate output, offset, carry or search-budget failures without relaxing the oracle."
},
{
"id": "terminal-performance.padded-fullscreen-redraw",
"title": "Fullscreen redraw padding does not stall terminal delivery",
@@ -345,6 +520,106 @@
],
"demotionRule": "Keep experimental or demote if adoption duplicates covered output, drops newer or unproven output, changes terminal ownership, or flakes without explanation."
},
{
"id": "terminal-session.io-failure-cleanup",
"title": "Native PTY I/O failures preserve termination ownership",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-runtime",
"layer": "provider-contract",
"surfaces": ["daemon PTY teardown"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local-daemon", "ssh-daemon", "paired-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local-daemon"],
"coverageNotes": "Real TerminalHost, Session, and subprocess wrapper with injected native I/O failures and mocked OS signals. Local non-daemon and SSH-relay implementations are unaffected; daemon consumers on SSH, WSL, paired runtimes, and mobile retain host-owned semantics. Live Linux/Windows/WSL and remote runs remain gaps. No git or folder-workspace assumptions. The fault-injection suite also runs with simulated darwin/linux/win32 platform branches; these do not constitute native OS coverage. Native macOS coverage now proves shell exit and PTY master-fd closure, input/output round trips, and teardown of a paused producer for both graceful and immediate cleanup. Windows single-close/job escalation and pre-listener output/status are fault-injected contracts.",
"motivatingLinks": ["docs/terminal-daemon-session-leak-investigation.md"],
"invariant": "I/O errors must not establish physical exit or disable termination of an owned PTY. Session and native handle disposal require the exit event.",
"oracle": "Inject write and resize failures, require graceful and forced signals to reach the native owner, keep producer resume available, suppress repeated failed I/O, deliver output and exit, and suppress signals after exit. Across 32 create/close cycles per failure, retain each session before exit and release its native handle and emulator exactly once afterwards. Mark physical exit before notifying listeners; reentrant kill/forceKill/signal from those listeners must never signal the retired PID. A native POSIX test performs input/output and resize, pauses the producer, injects each I/O failure, then gracefully or immediately closes 16 real shells; require ESRCH for each child PID and EBADF for each PTY master fd.",
"commands": [
"pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts",
"pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts src/main/daemon/pty-subprocess-handle-lifecycle.test.ts src/main/daemon/terminal-host-session-reaping-leak.test.ts src/main/daemon/terminal-host-teardown-recreate.test.ts src/main/daemon/terminal-session-teardown.test.ts src/main/daemon/session.test.ts",
"pnpm test src/main/daemon/pty-subprocess-io-failure-native.test.ts"
],
"testFiles": [
"src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts",
"src/main/daemon/pty-subprocess-handle-lifecycle.test.ts",
"src/main/daemon/terminal-host-session-reaping-leak.test.ts",
"src/main/daemon/terminal-host-teardown-recreate.test.ts",
"src/main/daemon/terminal-session-teardown.test.ts",
"src/main/daemon/session.test.ts",
"src/main/daemon/pty-subprocess-io-failure-native.test.ts"
],
"assertionRefs": [
{
"file": "src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts",
"assertions": [
"keeps graceful and forced termination available until physical exit",
"reaps every session and native handle across 32 failed-I/O create/close cycles",
"suppresses repeated native I/O failures while still delivering output and exit",
"blocks reentrant termination from an exit listener after I/O failure"
]
},
{
"file": "src/main/daemon/pty-subprocess-io-failure-native.test.ts",
"assertions": ["reaps real shells and master fds after %s failure (immediate=%s)"]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts src/main/daemon/pty-subprocess-handle-lifecycle.test.ts src/main/daemon/terminal-host-session-reaping-leak.test.ts src/main/daemon/terminal-host-teardown-recreate.test.ts src/main/daemon/terminal-session-teardown.test.ts src/main/daemon/session.test.ts",
"result": "passed",
"durationSeconds": 0.617,
"summary": "136 tests passed across six files; failed-I/O cycle tests cover 64 closures."
},
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts",
"result": "passed",
"durationSeconds": 3.71,
"summary": "32 tests passed with 4 Windows-only cases skipped; simulated macOS/Linux/Windows branches include 192 failed-I/O create/close cycles and exit-listener reentrancy."
},
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "pnpm test src/main/daemon/pty-subprocess-io-failure-native.test.ts",
"result": "passed",
"durationSeconds": 2.52,
"summary": "Four native cases pass across 16 real shells, including input/output, pause before teardown, confirmed PID absence, and closed PTY master fds."
}
],
"runtimeBudget": {
"p95Seconds": 10,
"scope": "focused daemon teardown contract tests"
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial deterministic local run; no soak history."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "All four original regression cases failed before the fix because native kill was never called; the unchanged cases passed after separating I/O failure from exit. Two additional output/flow-control cases also pass. Review added two failing exit-listener reentrancy cases; publishing physical exit before callbacks made them pass."
},
"performanceBudget": {
"required": true,
"evidence": "One boolean per PTY; no new timers, scans, retries, or subprocesses. Existing failed-I/O suppression remains. 192 closures under three simulated platform branches return session inventory to zero and dispose each emulator/native handle once."
},
"promotionCriteria": [
"Collect remaining native cross-platform evidence plus the standard soak history."
],
"knownGaps": [
"Fault injection proves a leak mechanism, not causality for the historical 427-session incident.",
"Real Linux/Windows/WSL, remote, startup-close, login-wrapper descendants, and multi-day load evidence remain outstanding. Native tests inject synchronous I/O errors; they do not model every asynchronous node-pty pipe failure.",
"No output throughput change or interactive latency benchmark is included."
],
"demotionRule": "Keep experimental; investigate any lost cleanup signal, premature exit, or unexplained flake."
},
{
"id": "cmd-j-tabs.host-qualified-candidate-ownership",
"title": "Cmd-J tab candidates retain execution-host ownership",
@@ -5902,7 +6177,7 @@
"invariant": "After a TUI exits or is killed, reveal, reattach, snapshot replay, or renderer remount must not deliver terminal-owned mouse or alternate-screen protocol bytes to the surviving shell. Recovery is an ordered output barrier in the daemon session data path: an OSC 133;D completing while the alternate screen is still active pauses the stream at that exact byte boundary, a fresh execution-host process inspection proves shell ownership, and on proof a mode reset is injected as in-stream output so every consumer converges by parsing the same bytes and the queued post-boundary shell output (the prompt) lands on the normal buffer. Snapshots are pure reads. Any failure — refuted proof, timeout, queue overflow, session death, disposal — flushes the queue unmodified, preserving incumbent behavior; later command or mode bytes revoke proof. Clean alternate-screen exits prove ownership asynchronously without pausing.",
"oracle": "Run one fixed child-TUI journey for normal exit and cleanup-free SIGKILL. Assert renderer and host normal-buffer/non-mouse state, host snapshot terminalOwner metadata, exact PTY writes with no post-exit mouse report, unrelated-pane survival, post-boundary prompt output preserved (normal exit), ordered proof invalidation, bounded settlement and bail-out flush, one inspection per unclean episode with zero scans for ordinary output, split-escape safety at every chunk boundary, and old/new client-host fallback parity.",
"commands": [
"pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/terminal-shell-lifecycle-scanner.test.ts src/main/daemon/terminal-shell-recovery-barrier.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/session.test.ts src/main/daemon/terminal-host-concurrent-create.test.ts src/main/daemon/daemon-pty-adapter.test.ts src/main/daemon/daemon-restore-scrollback-depth.test.ts src/main/daemon/terminal-checkpoint-serializer.test.ts src/main/providers/agent-foreground-process.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/mobile-subscribe-integration.test.ts src/main/runtime/rpc/terminal-multiplex-escape-tail.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-codex-queries.test.ts src/renderer/src/components/terminal-pane/pty-connection-reattach-mode-reset.test.ts src/renderer/src/components/terminal-pane/pty-connection-daemon-snapshot-replay.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-escape-tail.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/terminal-shell-lifecycle-scanner.test.ts src/main/daemon/terminal-shell-recovery-barrier.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/session.test.ts src/main/daemon/terminal-host-concurrent-create.test.ts src/main/daemon/daemon-pty-adapter.test.ts src/main/daemon/daemon-restore-scrollback-depth.test.ts src/main/daemon/terminal-checkpoint-serializer.test.ts src/main/providers/agent-foreground-process.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/mobile-subscribe-integration.test.ts src/main/runtime/rpc/terminal-multiplex-escape-tail.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-codex-queries.test.ts src/renderer/src/components/terminal-pane/pty-connection-reattach-mode-reset.test.ts src/renderer/src/components/terminal-pane/pty-connection-daemon-snapshot-replay.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-escape-tail.test.ts src/shared/terminal-partial-escape-tail.test.ts src/shared/terminal-partial-escape-tail.fuzz.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts tests/e2e/cross-version-wire/cross-version-terminal-wire.unit.test.ts --reporter=dot",
"pnpm exec electron-vite build --mode e2e",
"SKIP_BUILD=1 pnpm exec playwright test tests/e2e/terminal-hidden-child-tui-kill-mode-reset.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1"
@@ -5924,6 +6199,8 @@
"src/renderer/src/components/terminal-pane/pty-connection-reattach-mode-reset.test.ts",
"src/renderer/src/components/terminal-pane/pty-connection-daemon-snapshot-replay.test.ts",
"src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-escape-tail.test.ts",
"src/shared/terminal-partial-escape-tail.test.ts",
"src/shared/terminal-partial-escape-tail.fuzz.test.ts",
"tests/e2e/cross-version-wire/cross-version-terminal-wire.unit.test.ts",
"tests/e2e/terminal-hidden-child-tui-kill-mode-reset.spec.ts"
],
@@ -5945,6 +6222,13 @@
"a snapshot taken during a split escape keeps the pending tail intact and stale proof is revoked by the completing bytes"
]
},
{
"file": "src/shared/terminal-partial-escape-tail.fuzz.test.ts",
"assertions": [
"the pending tail this gate threads over the wire folds identically at every code-unit split of the combined stream, including boundaries landing inside oscEsc/stringEsc",
"the split sweep runs over an alphabet carrying CAN, SUB, doubled ESC inside OSC/DCS/SOS/PM/APC, BEL, C1 ST, NUL, DEL, intermediates, CJK, astral, and lone surrogates"
]
},
{
"file": "tests/e2e/terminal-hidden-child-tui-kill-mode-reset.spec.ts",
"assertions": [
@@ -13102,7 +13386,7 @@
"https://github.com/stablyai/orca/issues/13821",
"https://github.com/stablyai/orca/issues/14347"
],
"invariant": "Injected orchestration task prompts for recognized agent CLIs must send the prompt body inside one bracketed-paste frame, sanitize embedded ESC bytes, preserve chunk boundaries without losing the frame, and submit exactly once only after the agent can accept Enter. A successful orchestration.workerStart must durably record exactly one accepted and started turn; a swallowed Enter must fail with agent_prompt_stalled and never trigger a blind rescue Enter. Claude and Codex must emit a post-paste composer marker and then settle, or reach the bounded fallback first; every other agent retains the platform delay.",
"invariant": "Injected orchestration task prompts for recognized agent CLIs must send the prompt body inside one bracketed-paste frame, sanitize embedded ESC bytes, preserve chunk boundaries without losing the frame, and submit exactly once only after the agent can accept Enter. Local worker-start with supported observation must preserve an unobserved turn as start_unknown without revoking authority, closing questions, or triggering a rescue Enter; a worker report during observation must settle normally. Claude and Codex must emit a post-paste composer marker and then settle, or reach the bounded fallback first; every other agent retains the platform delay.",
"oracle": "Runtime tests assert the exact PTY write sequence, failure cleanup, Claude/Codex marker-gated multi-frame renders, and the legacy platform delay for every other configured agent. The candidate resets settlement on later frames, gives a late marker a fresh bounded window, and still submits once at the hard deadline if output never settles. The worker-start contract drives the production RPC through a delayed fake Codex composer and independently checks exact turn/Enter counts plus reopened SQLite Task, Dispatch, worker receipt, and mutation receipt state for accepted and swallowed outcomes. Other orchestration tests assert dispatch/coordinator use the agent prompt path; the live CLI harness covers long Codex-like framing.",
"commands": [
"pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
@@ -13152,7 +13436,8 @@
"file": "src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts",
"assertions": [
"delayed composer readiness produces exactly one submitted and started turn with no premature Enter and durable ready receipts",
"a swallowed Enter records agent_prompt_stalled across Task, Dispatch, worker, and mutation receipts without a rescue Enter"
"a swallowed Enter durably records start_unknown without a rescue Enter or capability revocation",
"early worker reports settle during observation, and outstanding questions survive observation uncertainty"
]
},
{
@@ -13553,6 +13838,7 @@
"invariant": "Starting a worker in the coordinator's current workspace must materialize one inactive terminal tab before worker-start returns, preserve coordinator focus, and remain exactly once after workspace re-entry. After an app update or restart, an exact live legacy worker must fence automatic provider resume, adopt its original PTY into its original background pane, retain readable output, and clear the resume record without spawning, writing, signalling, interrupting, replacing, or focusing the worker. A current-contract worker whose renderer graph identity is temporarily absent must retain its Dispatch capability and settle exactly once from exact hook-attested handle, pane, and process evidence; otherwise only an exact attested coordinator may take over. A worker_done caller may report success only after the owning runtime returns an explicit lifecycle verdict or authoritative reads prove that the exact Task, Dispatch, and worker report receipt settled the expected outcome. Federated terminal settlement must remain replay-eligible until the worker durably acknowledges it, and identical same-outcome retries must converge idempotently. Independently updated clients and worker servers must preserve the negotiated protocol: current peers use Run-home lifecycle settlement, while protocol v1/v2 peers retain their legacy completion path without receiving newer-only fields. A federated worker may accept only the authority defined by its negotiated protocol. An exact existing target workspace must receive a discoverable tab without stealing coordinator focus; if renderer reveal fails, worker-start must expose that the live worker remains background-only. Run and Dispatch checks must resolve through the caller's stable pane identity when a terminal handle is reminted, while a live handle outranks mismatched pane metadata. A nested worker's creator edge requires the current creator pane, process incarnation, and owning Run generation; reminting and rebinding that pane to another Run must remove the stale edge. Explicit legacy terminal inspection remains handle-scoped, and remote or headless worker presentation remains background-only.",
"oracle": "Drive Run create, Task create, and worker-start through production Electron runtimes with a deterministic Codex fixture. Require append-only ledgers with one still-live PID and no interruption, a visible inactive worker tab while the coordinator stays active, Run delivery through stable pane identity, and stable PTY/incarnation, tab, leaf, worktree, Task, and Dispatch across workspace re-entry. In a restart journey, retain the original daemon PTY and PID, remove renderer ownership, retain sleeping-session evidence, mark the Dispatch legacy, relaunch, and require exact inactive tab adoption, readable ACK output, cleared resume state, one spawn, and no resume argv or Conversation interrupted text after another workspace round trip. The service oracle removes renderer lookup identity from current-contract callers while retaining real restored-PTY and hook commitments, replays authenticated completion and takeover across fresh runtimes, and requires one Task, Dispatch, terminal authority, message, mutation, ordinary-mail delivery, remote process fencing, and unchanged fixture marker bytes while foreign pane evidence remains rejected. Unit tests separately remint a creator pane and process from Run A into Run B, require the nested Run A worker to fall back to its current coordinator, require indexed query plans, and bound 300 Task reads with 50,000 retained Runs. They also assert authority-specific legacy affordances, exact identity and owner matching, retained-output fallback, pane-stable routing, federated non-activation, and SSH fallback parity.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 npx vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration.test.ts src/cli/handlers/orchestration-check-identity.test.ts src/cli/handlers/orchestration-worker-cli.test.ts src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts src/main/ssh/ssh-remote-orca-cli.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration-lifecycle-rejection.test.ts src/cli/handlers/orchestration-lifecycle-json-rejection.test.ts src/cli/handlers/orchestration-migration.test.ts",
@@ -13567,6 +13853,7 @@
"pnpm run build:cli && SKIP_BUILD=1 pnpm exec playwright test tests/e2e/orchestration-worker-settlement-release-cli.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1"
],
"testFiles": [
"src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts",
"src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts",
"src/main/runtime/orchestration/formatter.test.ts",
"src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts",
@@ -13590,6 +13877,13 @@
"tests/e2e/orchestration-worker-settlement-release-cli.spec.ts"
],
"assertionRefs": [
{
"file": "src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts",
"assertions": [
"replays the coordinator instruction and takes its ack after the app restarts",
"files loopback mail once under the local Dispatch Run without replacing its owner"
]
},
{
"file": "src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts",
"assertions": [
@@ -14088,7 +14382,7 @@
"providers": ["local", "daemon", "ssh", "wsl", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "ssh"],
"coverageNotes": "Deterministic service tests cover release-versus-reuse ordering, transactional retain and takeover cancellation, exact host/pane/process identity, dead external/user-owned/transferred/stopped/abandoned reconciliation, host-partition persistence and legacy retirement replay with an absent web-terminal layout map, conservative unknown provider and legacy metadata handling, immutable transcript and bounded terminal archives, mutation restart, reset cleanup, replay idempotency, and 50-resource accounting. A macOS Electron journey invokes the freshly compiled worker-release CLI after the worker process disappears, then independently checks released SQLite state and coordinator liveness. Injected inventories cover local and SSH provider routing; live SSH, WSL, Windows, paired-runtime, and provider-close lost-ack journeys remain explicit gaps.",
"coverageNotes": "New phones explicitly report terminal takeover on real user sends, throttled per owning client and handle. Host byte lanes perform zero orchestration SQL work; local and injected SSH report tests fence release. Phones predating this build do not fence release. Deterministic service tests cover release-versus-reuse ordering, transactional retain and takeover cancellation, exact host/pane/process identity, dead external/user-owned/transferred/stopped/abandoned reconciliation, host-partition persistence and legacy retirement replay with an absent web-terminal layout map, conservative unknown provider and legacy metadata handling, immutable transcript and bounded terminal archives, mutation restart, reset cleanup, replay idempotency, and 50-resource accounting. A macOS Electron journey invokes the freshly compiled worker-release CLI after the worker process disappears, then independently checks released SQLite state and coordinator liveness. Injected inventories cover local and SSH provider routing; live SSH, WSL, Windows, paired-runtime, and provider-close lost-ack journeys remain explicit gaps.",
"motivatingLinks": [
"https://github.com/stablyai/orca/pull/12355",
"https://github.com/stablyai/orca/issues/13860",
@@ -14099,6 +14393,8 @@
"invariant": "A settled Dispatch may close only its one coordinator-created terminal lease. Explicit reuse, real user input, retain, identity or host change, ambiguity, and another resource for the same exact host/pane/process must fence closure. Once the authoritative owning provider positively excludes the resource's exact immutable process incarnation, even an external, user-owned, or transferred dead resource must converge to released without any process close. Unknown host scope, missing incarnation metadata, or unavailable inventory must remain retained. Exact terminal-close persistence must settle when a host partition omits renderer-owned layout state. Output preservation and the requested-to-releasing transition are atomic, archives remain readable without the provider file, retries resume idempotently, and orchestration reset removes archive and authority state.",
"oracle": "Record release intent for a settled owner, attempt exact reuse before close, and require worker-start to fail with terminal_release_in_progress while the terminal stays open; then release the original owner exactly once. Race retain and real user input against a controlled archive promise and require no committed archive or close. Rebase a closed web-terminal host partition without terminalLayoutsByTabId and require the persistence write to complete while preserving host-authoritative membership; replay a valid legacy retirement under the same omission and require exact membership removal plus revision advancement. For retained external, user-owned, transferred, stopped, and abandoned resources, run one fresh inventory against the exact local/WSL or SSH provider: an exact live incarnation and every unknown inventory shape stay retained, while positive absence atomically sets ownership_state and release_state to released with processAction none and zero closeTerminal calls. Change host or process identity and inject duplicate resource evidence to require retention. Freeze a structured transcript, delete its source file, and require archived worker-read to return the same bounded redacted messages. Restart a pending mutation, reset orchestration state, and create 50 resources while asserting replay convergence, zero orphan rows, two-query worker listing, and no unrelated close.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts src/main/runtime/orca-runtime-terminal-handle-incarnation.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts",
"ORCA_BACKGROUND_LAUNCH=1 mobile/node_modules/.bin/vitest run --config mobile/vitest.config.ts mobile/src/session/mobile-worker-takeover-send-sites.test.ts mobile/src/terminal/worker-terminal-takeover-report.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/pty-inventory-liveness-verdict.test.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
@@ -14107,6 +14403,9 @@
"pnpm run build:cli && SKIP_BUILD=1 pnpm exec playwright test tests/e2e/orchestration-worker-settlement-release-cli.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1"
],
"testFiles": [
"src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts",
"mobile/src/session/mobile-worker-takeover-send-sites.test.ts",
"mobile/src/terminal/worker-terminal-takeover-report.test.ts",
"src/main/runtime/pty-inventory-liveness-verdict.test.ts",
"src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts",
"src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts",
@@ -14119,6 +14418,20 @@
"tests/e2e/orchestration-worker-settlement-release-cli.spec.ts"
],
"assertionRefs": [
{
"file": "src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts",
"assertions": [
"a handle-addressed phone report fences %s worker release",
"mobile %s bytes do no orchestration database work"
]
},
{
"file": "mobile/src/session/mobile-worker-takeover-send-sites.test.ts",
"assertions": [
"%s reports on its send target once per handle per 30 seconds",
"%s never reports takeover"
]
},
{
"file": "src/main/runtime/pty-inventory-liveness-verdict.test.ts",
"assertions": [
@@ -18378,7 +18691,7 @@
"providers": ["ssh"],
"coveredPlatforms": ["macos", "linux"],
"coveredProviders": ["ssh"],
"coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075. Deterministic remote Codex fixture validation passed three normal restores and three forced reconnects with zero retries on merged main plus the replay probe correction (run 34050117471). The original forced-reconnect probe missed nonempty replay returned in pty:spawn reattach replies. Routine coverage now includes both modes by default; real Codex service execution remains opt-in.",
"coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075. Deterministic remote Codex fixture validation passed three normal restores and three forced reconnects with zero retries on merged main plus the replay probe correction (run 34050117471). The original forced-reconnect probe missed nonempty replay returned in pty:spawn reattach replies. Routine coverage now includes both modes by default; real Codex service execution remains opt-in. The added five-pane input spec passed in Linux CI run 34033353595, and diagnostic run 34034754815 reproduced real input loss during scrollback replay; application fix #19075 (98b0c329ff3) has since merged and this spec now guards it.",
"motivatingLinks": [
"https://github.com/stablyai/orca/issues/18018",
"https://github.com/stablyai/orca/pull/18546",
@@ -18396,7 +18709,8 @@
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1",
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10",
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-codex-display-artifacts-repro.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1",
"pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts"
"pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts",
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1"
],
"testFiles": [
"tests/e2e/ssh-docker-transport-drop-recovery.spec.ts",
@@ -18408,7 +18722,8 @@
"tests/e2e/helpers/electron-process-shutdown.unit.test.ts",
"tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts",
"tests/e2e/ssh-codex-display-artifacts-repro.spec.ts",
"tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts"
"tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts",
"tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts"
],
"assertionRefs": [
{
@@ -18460,6 +18775,12 @@
"five flooding SSH panes remain below unchanged 2500ms soft and 5000ms hard freeze budgets during bulk reopen and two double-animation-frame view changes"
]
},
{
"file": "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts",
"assertions": [
"five distinct SSH PTYs acknowledge actual keyboard input after two rendered hide/reopen cycles while all five producers flood"
]
},
{
"file": "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts",
"assertions": [
@@ -18508,7 +18829,7 @@
},
"flakeHistory": {
"status": "flaky",
"evidence": "Baseline: ten enabled tests passed, two fixme skipped, worker teardown timed out (7.3m). After pipe cleanup: ten passed and worker exited cleanly (5.2m); two half-open repeats passed (1.7m). The formerly skipped thaw-input case failed before its recovered-authority wait and passed 1+3 executions afterward (56.9s + 2.6m). Flood failed both its original input oracle and a strengthened producer-completion oracle after recovery."
"evidence": "Baseline: ten enabled tests passed, two fixme skipped, worker teardown timed out (7.3m). After pipe cleanup: ten passed and worker exited cleanly (5.2m); two half-open repeats passed (1.7m). The formerly skipped thaw-input case failed before its recovered-authority wait and passed 1+3 executions afterward (56.9s + 2.6m). Flood failed both its original input oracle and a strengthened producer-completion oracle after recovery. Five-pane input diagnostics additionally failed 1/5 in run 34034754815: the intended focused PTY emitted the full input, the replay guard discarded 31 characters, and the remote ACK contained exactly the remaining suffix. PR #19075 addresses that application bug; passing repetitions alone do not establish its resolution."
},
"redGreenEvidence": {
"status": "partial",
@@ -18524,6 +18845,7 @@
"Collect CI runtime and flake history plus product red/green evidence before blocking."
],
"knownGaps": [
"Five-pane simultaneous flood input was reproduced as a real application bug in run 34034754815 (replay discarded the first 31 characters of correctly focused keyboard input); fix #19075 (98b0c329ff3) merged and the spec now guards it, but the retries: 0 Docker SSH lane is the only repetition evidence against the merged fix so far. Freeze performance coverage was restored separately in #19081, with its isolated headless timer-lag outlier still documented.",
"The disconnected 48MB flood still loses its relay channel: original post-flood input marker failed in 60s, and waiting for the finite producer completion marker failed in 120s. It remains an explicit #18018 fixme reproduction; frozen-host input is re-enabled after four successful runs.",
"Linux headed CI covers the bulk-open freeze reproduction; Windows clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not covered by that result.",
"Some legacy assertions inspect terminal serialization or backing state rather than rendered DOM; no blanket visual coverage claim.",
@@ -1,6 +1,6 @@
export async function configureRendererScaleFixture(page, options, repoPath) {
return page.evaluate(
({ agentsPerWorktree, lineageDepth, repoPath }) => {
({ agentsPerWorktree, subagentsPerAgent, lineageDepth, repoPath }) => {
const store = window.__store
if (!store) {
throw new Error('window.__store is not available')
@@ -98,7 +98,18 @@ export async function configureRendererScaleFixture(page, options, repoPath) {
{
state: 'working',
prompt: `Idle CPU agent ${worktreeIndex + 1}.${agentIndex + 1}`,
agentType
agentType,
...(subagentsPerAgent > 0
? {
subagents: Array.from({ length: subagentsPerAgent }, (_, index) => ({
id: `child-${index}`,
state: 'working',
startedAt: fixtureNow,
agentType,
description: `Subagent ${worktreeIndex + 1}.${agentIndex + 1}.${index + 1}`
}))
}
: {})
},
agentType,
{ updatedAt: fixtureNow, stateStartedAt: fixtureNow },
@@ -116,10 +127,16 @@ export async function configureRendererScaleFixture(page, options, repoPath) {
expandedLineageGroups: lineageParentIds.size,
agentsPerWorktree,
seededAgentRows,
seededSubagentRows: seededAgentRows * subagentsPerAgent,
orderedWorktreeIds: worktrees.map((worktree) => worktree.id)
}
},
{ agentsPerWorktree: options.agentsPerWorktree, lineageDepth: options.lineageDepth, repoPath }
{
agentsPerWorktree: options.agentsPerWorktree,
subagentsPerAgent: options.subagentsPerAgent ?? 0,
lineageDepth: options.lineageDepth,
repoPath
}
)
}
@@ -129,6 +129,7 @@ for (const count of [36, 50, 250]) {
const issues = makeJiraIssues(count)
const before = () =>
[...issues]
// oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Baseline measures per-comparison setup against a reused collator.
.sort((a, b) => a.key.localeCompare(b.key, undefined, { numeric: true }))
.map((issue) => issue.key)
const after = () => sortJiraIssues(issues, 'key', 'asc').map((issue) => issue.key)
@@ -142,6 +143,7 @@ for (const count of [36, 50, 250]) {
for (const count of [10, 50, 250]) {
const values = makeBaseSensitivityValues(count)
const before = () =>
// oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Baseline measures per-comparison setup against a reused collator.
[...values].sort((a, b) => a.localeCompare(b, undefined, { sensitivity: 'base' }))
const after = () => [...values].sort(compareBaseSensitivityLocaleText)
assertSameOrder(before, after, `base ${count}`)
@@ -117,7 +117,7 @@ function blockContent(message: NativeChatMessage): string {
if (block.type === 'tool-result') {
return block.output
}
return block.path ?? block.url ?? block.alt ?? ''
return block.type === 'image-ref' ? (block.path ?? block.url ?? block.alt ?? '') : block.groupId
}
function messageWeight(message: NativeChatMessage, content: string): number {
@@ -168,9 +168,10 @@ describe('orchestration kernel', () => {
expect(kernel).toContain(
'`projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv'
)
expect(kernel).toContain(
'An `inspect` `nextAction` on a `live` row with `attention.requiresAction` false is informational, not a command to re-run: keep waiting with `check --wait`'
)
// Unverifiable workers can still owe release; the guide must explain the action itself.
expect(kernel).toContain('A `none` `nextAction` has no argv to run')
expect(kernel).toContain('read `liveness.reason` and keep waiting with `check --wait`')
expect(kernel).toContain('Absence never earns an argv; settlement and pending work still do')
expect(kernel).toContain('choose `worker-stop` or `worker-abandon`')
})
@@ -168,6 +168,9 @@ describe('PR E2E gate contract', () => {
expect(changedRun.env.TEST_FILES_JSON).toBe('${{ inputs.test_files }}')
expect(changedRun.run).toContain('. != "tests/e2e/ssh-startup-exec-readiness.spec.ts"')
expect(changedRun.run).toContain('. != "tests/e2e/paired-startup-exec-readiness.spec.ts"')
expect(changedRun.run).toContain(
'. != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts"'
)
expect(changedRun.run).toContain('. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"')
expect(changedRun.run).toContain('if [ "${#TEST_FILES[@]}" -eq 0 ]')
expect(changedRun.run).toContain('grep -l \'@headful\' "${TEST_FILES[@]}"')
+1
View File
@@ -63,6 +63,7 @@ const result = spawnSync(
'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts',
'tests/e2e/ssh-cold-activation-restore.spec.ts',
'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts',
'tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts',
'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts',
'tests/e2e/ssh-docker-half-open-link.spec.ts',
'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts',
+1 -1
View File
@@ -34,7 +34,7 @@ src/main/runtime/orca-runtime-create-terminal-side-effect-command-code-detector.
src/main/runtime/orca-runtime-create-terminal.ts
src/main/runtime/orca-runtime-deliver-pending-messages.ts
src/main/runtime/orca-runtime-emit-daemon-pty-transient-fact.ts
src/main/runtime/orca-runtime-fence-automation-owner.ts
src/main/runtime/orca-runtime-automation-operations.ts
src/main/runtime/orca-runtime-file-commands.ts
src/main/runtime/orca-runtime-fit-override-listeners.ts
src/main/runtime/orca-runtime-focus-terminal.ts
+4 -4
View File
@@ -1,5 +1,5 @@
<svg xmlns="http://www.w3.org/2000/svg" width="106" height="20" role="img" aria-label="downloads: 42m">
<title>downloads: 42m</title>
<svg xmlns="http://www.w3.org/2000/svg" width="106" height="20" role="img" aria-label="downloads: 44m">
<title>downloads: 44m</title>
<linearGradient id="s" x2="0" y2="100%">
<stop offset="0" stop-color="#bbb" stop-opacity=".1"/>
<stop offset="1" stop-opacity=".1"/>
@@ -15,7 +15,7 @@
<g fill="#fff" text-anchor="middle" font-family="Verdana,Geneva,DejaVu Sans,sans-serif" text-rendering="geometricPrecision" font-size="11">
<text x="37" y="15" fill="#010101" fill-opacity=".3">downloads</text>
<text x="37" y="14">downloads</text>
<text x="90" y="15" fill="#010101" fill-opacity=".3">42m</text>
<text x="90" y="14">42m</text>
<text x="90" y="15" fill="#010101" fill-opacity=".3">44m</text>
<text x="90" y="14">44m</text>
</g>
</svg>

Before

Width:  |  Height:  |  Size: 935 B

After

Width:  |  Height:  |  Size: 935 B

@@ -87,13 +87,25 @@ bundled prototype, the fixture without seeded agents fell from 8,518 listeners
to 1,218; with 100 visible agent rows the candidate mounted 1,618. Compare
against the census in "Baseline on `main`", which the harness reports directly.
### Share working-spinner phase without per-element animation queries
### Share working-spinner phase without synchronous mount queries
Working rows keep the existing compositor-driven CSS animation and shared
visual phase. Each mount derives one negative animation delay from the document
timeline instead of querying `getAnimations()` and mutating the animation start
time. This removes per-row Web Animations setup from dense status transitions
without adding a JavaScript animation clock.
visual phase. `animationstart` anchors each animation to document time zero.
Deferring the animation query until that event avoids a synchronous style flush
at each mount and restores the shared phase after `display:none` or a motion
preference change. A negative mount-time delay cannot preserve that phase after
an animation restarts.
### Bound spinner animation overhead
Working rings keep compositor-driven CSS animation, but repeat the animation
once per day rather than once per second. The same 12 steps per second now
avoid recurring React animation-iteration dispatch. The existing stationary
wrapper and ring rendering stay unchanged. Offscreen containment was evaluated
and rejected after a pixel regression at low zoom on 1x displays.
The history, isolated measurements, full-app workspace/agent/subagent benchmark,
and limitations are documented in [Spinner rendering performance](./spinner-rendering-performance.md).
### Fold a burst in event order
@@ -0,0 +1,203 @@
# Spinner rendering performance
## ELI5
Imagine a wheel that tells the front desk every time it completes a lap. The
front desk is also handling your typing. CSS already turns the wheel for us,
but React still receives its once-per-second lap notifications.
We put a day's worth of laps into one animation. The wheel moves at the same
speed, while sending one lap notification a day. Drawing visible wheels still
costs something. This removes recurring bookkeeping from the input thread; it
does not make rendering or the rest of Orca free.
## How this builds on earlier changes
| Change | What it achieved | Remaining cost |
| ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- |
| [#9380](https://github.com/stablyai/orca/pull/9380): shared JavaScript clock | Reduced frame-pipeline CPU in the original one-agent measurement | Wrote each spinner's style 12 times per second on the input thread |
| [#12359](https://github.com/stablyai/orca/pull/12359): compositor CSS rotation | Removed those recurring JavaScript style writes; fixed the reported typing regression | React still receives CSS iteration events |
| [#13987](https://github.com/stablyai/orca/pull/13987): synchronize on animationstart | Avoided a synchronous style query at every mount | Steady-state animation overhead stayed the same |
| This change | Preserves both later fixes and removes almost all iteration boundaries | Compositing, other app work, mount/reveal work, and a daily iteration boundary remain |
The historical measurements in #12359 reported 41 rings causing about 490 style
writes per second, with typing input-delay p90 of 363 ms versus 19 ms when those
writes stopped. Those are historical production measurements, not numbers from
this benchmark or a direct comparison with today's app.
## Implementation
The production change is entirely in CSS. `AgentWorkingSpinner`, its callers,
markup, border, animation-start handler, and reduced-motion behavior stay the
same. No DOM node, pseudo-element, containment boundary, timer, observer, or
JavaScript animation loop is added.
The transform travels 86,400 turns in 86,400 seconds with 1,036,800 steps: exactly
one revolution and 12 steps per second. `animationstart` sets `startTime = 0` as
before, preserving shared phase after mount and animation restart. The step
count is a timing-function parameter, not a million-entry keyframe list.
React installs delegated `animationiteration` listeners even when the component
has no iteration handler. A native 2.2-second trace of 200 isolated rings counted
400 iteration events and 800 JavaScript calls before the change, versus zero of
either with the long cycle. That trace installed no animation-event listener.
These are event dispatches, not component rerenders or 400 separate OS wakeups.
## Full-app benchmark
The opt-in Playwright benchmark launches a fresh, hidden Orca app for each
scenario. It creates real Git workspaces and seeds working statuses through the
existing renderer fixture, including in-process subagent data. It renders the
normal sidebar, virtualizer, lineage, agent rows, tabs, and terminal.
| Scenario | Git workspaces | Root agents | Subagents | Mounted / visible rings | Layout |
| ------------- | -------------: | ----------: | --------: | ----------------------: | -------------------------------------------- |
| `one-agent` | 1 | 1 | 0 | 3 / 3 | One working agent |
| `one-family` | 1 | 2 | 4 | 8 / 8 | All family rows expanded |
| `200-flat` | 200 | 400 | 800 | 162 / 15 | Normal virtualization; 23 workspaces mounted |
| `200-lineage` | 200 | 400 | 800 | 1,401 / 15 | Expanded lineage; all 200 workspaces mounted |
Measurement-only styles switch between the original one-second cycle and the
new long cycle on the same elements. The real React root, callers, status data,
and app stay the same. The reported run alternates A/B and B/A, with four
ten-second CPU samples per variant after warmup. CPU samples use cumulative
Electron process CPU and CDP main-thread task/script/style/layout metrics. No
renderer polling, screenshots, or benchmark iteration listeners run during
those CPU windows. No samples are discarded.
Typing is measured separately using the existing paced terminal-typing probe:
64 keys at 113 ms cadence, twice per variant, after two seconds of warmup with
status traffic. Status updates arrive in groups of up to eight every 200 ms.
Keys pass through the DOM, real PTY, and xterm. A sidecar timestamps arrival at
the PTY, and a bounded terminal-buffer scan observes each echo. Missing input
or echoes fail the benchmark. Echo measurements include the 10 ms scan interval;
they do not measure native display presentation. Native animation traces also
run separately from CPU and typing samples.
The statuses are deterministic test data, not hundreds of paid model sessions.
The test exercises UI cost under agent-status traffic, not the compute or network
cost of model inference, SSH traffic, or hundreds of streaming PTYs.
## Results
CPU values are medians of four samples. "CPU ms/s" means milliseconds of
processor time used in one wall-clock second: 100 ms/s is about 10% of one CPU
core. Renderer + GPU-process CPU includes their other app work and CPU used by
the graphics process; it is not GPU hardware utilization or whole-machine CPU.
The main thread handles input and is included in renderer CPU, not extra work.
Echo p90 means 90% of sampled keys were observed within that time; ranges show
the two runs, not confidence intervals. No keys or echoes were missing.
| Scenario | Renderer + GPU CPU ms/s, old → new | Main-thread ms/s, old → new | Echo p90 ms, old → new |
| ------------- | ---------------------------------: | --------------------------: | ---------------------- |
| `one-agent` | 37.0 → 38.2 | 5.4 → 3.5 | 19 → 18–19 |
| `one-family` | 46.2 → 44.6 | 7.8 → 4.4 | 17–19 → 18–19 |
| `200-flat` | 141.6 → 122.8 | 28.2 → 16.3 | 26–28 → 26–28 |
| `200-lineage` | 324.6 → 295.6 | 140.8 → 70.0 | 159–239 → 93–160 |
The consistent gain is less main-thread work: about 35%, 43%, 42%, and 50%
less in these four scenarios. Native 2.2-second traces counted 6, 16, 324, and
2,802 iteration events before, and zero in each new variant, without adding an
iteration listener. That avoided work also exists in Orca itself, independently
of the isolated fixture and CPU noise.
Total CPU was roughly unchanged in the one-worktree cases. In this run it fell
13% with normal virtualization and 9% with expanded lineage; seven of eight
paired large-case CPU samples favored the change. These percentages are not
universal: a shorter three-variant ablation measured flat-list CPU at 89.0 ms/s before and
108.0 ms/s with the long cycle, while main-thread time still fell from 26.3 to
17.4 ms/s. The repeatable main-thread reduction is stronger evidence than a
single total-CPU percentage.
Typing was similar in the small and flat-list cases. Expanded-lineage echo p90
improved in the final run, but a shorter ablation had similar before/after
latencies. No general typing speedup or statistical non-regression guarantee
is established by these short experiments.
### All CPU samples
Values are rounded to one decimal and listed by round, with no outliers removed.
The first new small-case samples were higher than their paired baselines; they
remain included. CPU and typing were sampled separately.
| Scenario | Version | Renderer + GPU CPU ms/s | Main-thread ms/s |
| ------------- | ------- | -------------------------- | -------------------------- |
| `one-agent` | Old | 37.8, 36.2, 26.2, 39.6 | 6.5, 5.2, 4.8, 5.7 |
| `one-agent` | New | 53.3, 35.8, 37.5, 39.0 | 6.7, 2.8, 3.0, 4.1 |
| `one-family` | Old | 46.3, 46.1, 47.9, 44.6 | 7.8, 7.6, 9.8, 7.7 |
| `one-family` | New | 53.8, 45.4, 42.5, 43.8 | 6.8, 4.5, 3.1, 4.3 |
| `200-flat` | Old | 142.4, 140.8, 147.0, 136.5 | 28.3, 28.1, 32.2, 27.0 |
| `200-flat` | New | 122.9, 97.2, 122.7, 126.7 | 19.2, 10.7, 16.6, 16.0 |
| `200-lineage` | Old | 317.9, 385.2, 315.9, 331.2 | 134.4, 159.6, 133.9, 147.3 |
| `200-lineage` | New | 318.6, 256.9, 296.5, 294.7 | 89.2, 60.8, 71.6, 68.3 |
## Reproduce
```sh
ORCA_BACKGROUND_LAUNCH=1 pnpm bench:spinners --sample-ms=5000
ORCA_BACKGROUND_LAUNCH=1 pnpm bench:spinners --verify-only --scale-factor=1
ORCA_BACKGROUND_LAUNCH=1 pnpm bench:spinners --verify-only --scale-factor=2
ORCA_BACKGROUND_LAUNCH=1 ORCA_SPINNER_BENCH=1 ORCA_SPINNER_KEYS=64 \
pnpm test:e2e spinner-workspace-perf.spec.ts --workers=1
```
The full-app command rebuilds in `e2e` mode. For a fresh build already made with
`pnpm exec electron-vite build --mode e2e`, `SKIP_BUILD=1` reuses it. Do not reuse
an old launch-policy build. `ORCA_SPINNER_SAMPLE_MS`, `ORCA_SPINNER_ROUNDS`,
`ORCA_SPINNER_KEYS`, `ORCA_SPINNER_KEY_CADENCE_MS`, `ORCA_SPINNER_VARIANTS`, and
`ORCA_SPINNER_OUTPUT` control the experiment. `ORCA_SPINNER_CPU=0` repeats only
typing; `--grep one-agent` selects one scenario. Reports, native traces, typing
sidecars, and CDP screenshots are written under `.bench-fixtures/`. Run one
benchmark at a time, without concurrent builds or tests.
The optional `contained` variant retains the rejected offscreen experiment for
ablation. It adds `content-visibility:auto` to the existing wrapper through
measurement-only styles. It is not enabled in production or the default
benchmark comparison.
## Visual and behavioral checks
Both 1x and 2x display-density checks passed 720 ring comparisons each: 6/8 px
rings, light/dark themes, supported zoom extremes, all 12 phases, long elapsed
times, and the daily wrap. The comparison pauses each animation and sets its
`currentTime`, so the long-elapsed and daily-wrap cases exercise the deterministic
style path rather than a running compositor animation. Against that path the
tolerance is one channel level for floating-point antialias rounding. A running
animation at multi-hour ages can differ by a few channels on the ring edge — a
fraction-of-a-pixel antialias difference at large accumulated angles, not a phase
or shape change. Checks also cover shared phase, reduced motion, initial offscreen
reveal, repeated scroll-away/reveal, and `display:none` restoration.
## Limits and rejected approaches
Adding `content-visibility:auto` to the existing stationary wrapper saved more
CPU at large mounted counts, but a 1x display check found a one-pixel shift at
the minimum UI zoom. That containment change is excluded. A previous
pseudo-element version also regressed typing latency in the virtualized list.
Neither prototype's CPU or typing numbers describe the final patch.
An initial typing run used a 100 ms key cadence, which can repeatedly align with
200 ms status bursts. Follow-up runs use 113 ms, more keys, and two seconds of
warmup under status traffic. This reduces timing bias; it does not excuse a
regression. CPU measurements run separately and do not depend on key cadence.
An early isolated test suggested a 31% process-CPU reduction that a longer audit
did not reproduce. The longer isolated audit measured original 104.04 versus
long-cycle 92.32 CPU ms/s, and main-thread 10.08 versus 0.24 ms/s. A fixture with
every ring far offscreen and containment enabled could also approach idle; that
is not representative of Orca with visible animations. Neither result justifies
claiming "free spinners" or a universal CPU percentage. Virtualized, unmounted
rows already cost nothing, and this patch does not add offscreen culling.
All local measurements use an Apple M4 (10 cores), macOS, Electron 43.4.1 /
Chromium 150.0.7871.224. Native windows stay hidden and unfocused;
benchmark-only settings disable background throttling to exercise the frame
pipeline. These are not visible-window power measurements. No battery benefit
is established. Linux/Windows need their own runtime measurements. The
renderer-only change does not alter SSH execution, wire data, status semantics,
Git operations, or folder-workspace ownership.
Animated PNGs, masks, layer promotion, CSS sprites, individual `rotate`, and
containment on the rotating element were also explored. Shared images added
raster work and regressed the single-ring case; sprites reintroduced per-frame
style work. They did not meet the appearance and responsiveness requirements.
+46
View File
@@ -0,0 +1,46 @@
# Structured worktree status validation
Validated on September 7, 2026 in a background Electron dev instance of
`pr19217-review-r2`, based on `ce1024096b` with the source-adapter refactor.
CDP app identity confirmed the checkout; screenshots show the full hidden renderer.
The command output is the real `orca worktree ps --json` response reduced to status,
agent state, provider, and pane key for readability.
## Functional correctness
A real Codex structured session appeared as `working` in `worktree.ps` while the
sidebar showed working. Closing its chat tab removed that exact session's row and
returned the worktree to `active`. A different completed chat remained present,
confirming that closure removed only the selected session.
- [Working: CLI and sidebar](working.png)
- [Closed: CLI and sidebar](closed.png)
The disappearing session is `codex_40677067_f492_4d7d_86dd_ec566ede04c3`.
The host's held-session roster controls eligibility; its retained broadcast cache
is history, not a roster. Failed eviction intentionally keeps an entry for retry.
## Architecture
PTY reconciliation and process admission belong to the PTY source adapter.
Structured input comes from the current host's held-session projections. One
admitted collection feeds row shaping and worktree aggregation, with no structured
boolean bypass. PTY hooks and retained reports still arrive independently, so their
precedence and conservative remote evidence rules remain necessary. No second
persistent status store or provider polling was introduced.
## Validation and limits
Independent final review found no proven issues. Runtime, host lifecycle, status
feed and source-admission suites passed: 1,344 tests, one skipped. Node typecheck,
targeted lint and diff checks passed. Ablating the runtime call to enumerate
retained history caused the executable call-site test to fail with two rows where
one was expected; restoring the live accessor passed both call-site tests.
Live screenshots prove Codex working and closure on macOS. Claude provider turns,
approval/input states, live Windows/Linux/WSL/SSH/relay/mobile scenarios and
release-scale latency/heap measurements remain unverified. Existing tests cover
remote/WSL evidence, monitoring precedence and lifecycle cases. The existing
30-minute freshness rule and CLI activity timestamps are preserved; complete
CLI/sidebar timing parity is not claimed. The wire keeps its existing row shape
and status vocabulary; mobile receives the new rows without a new opcode.
Binary file not shown.

After

Width:  |  Height:  |  Size: 103 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 109 KiB

@@ -1,3 +1,8 @@
// Takeover RPCs have their own send-site integration tests; these fixtures script PTY acknowledgements.
vi.mock('../terminal/worker-terminal-takeover-report', () => ({
reportWorkerTerminalUserInput: vi.fn()
}))
import { createElement } from 'react'
import { act, create, type ReactTestRenderer } from 'react-test-renderer'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
@@ -309,10 +309,13 @@ describe('typeMobileNativeChatCommandWithOutcome', () => {
await expect(result).resolves.toBe('accepted')
expect(
vi.mocked(client.sendRequest).mock.calls.map((call) => {
const params = call[1] as { text: string; enter: boolean }
return { text: params.text, enter: params.enter }
})
vi
.mocked(client.sendRequest)
.mock.calls.filter(([method]) => method === 'terminal.send')
.map((call) => {
const params = call[1] as { text: string; enter: boolean }
return { text: params.text, enter: params.enter }
})
).toEqual(
['\x15', '/', 'm', 'o', 'd', 'e', 'l', '\r'].map((text) => ({
text,
@@ -338,7 +341,12 @@ describe('typeMobileNativeChatCommandWithOutcome', () => {
await vi.runAllTimersAsync()
await result
const params = vi.mocked(client.sendRequest).mock.calls.map((call) => call[1]) as Array<{
// Why the filter: an accepted send also fires the unawaited takeover report, which is not a
// terminal.send and carries no draft.
const params = vi
.mocked(client.sendRequest)
.mock.calls.filter((call) => call[0] === 'terminal.send')
.map((call) => call[1]) as Array<{
text: string
resolvedLaunchDraft?: { text: string; createdAt: number }
}>
@@ -1,3 +1,4 @@
import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report'
import type { RpcClient } from '../transport/rpc-client'
import { isRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity'
import { isLogicalClientCutoverError } from '../transport/stable-logical-rpc-client'
@@ -64,7 +65,11 @@ export async function sendMobileNativeChatMessageWithOutcome(
// pins the composer for twice as long.
{ timeoutMs, budgetSpansConnect: true }
)
return isTerminalSendRpcAccepted(response) ? 'accepted' : 'rejected'
if (!isTerminalSendRpcAccepted(response)) {
return 'rejected'
}
reportWorkerTerminalUserInput(args.client, args.terminal)
return 'accepted'
} catch (error) {
// Why: a logical relay↔direct cutover rejects the in-flight send without
// knowing whether its frame reached the wire (the desktop may have delivered
@@ -66,11 +66,11 @@ const HEAD_MAIN_HOOK_SHA256 = '10071240ef9edafc2b9c8bed73be83dceaf7828e3b29f17da
const HEAD_HOOK_BINDING_SHA256 = '1dadb8c3dc0573ea20659ce7251629669e618dd0effaeac3a4536b29c2e865a1'
const HEAD_CALLBACK_IDENTITY_SHA256 =
'2a9e4825df007f6ef53b81aa5004991d6318eee7507b44d625c07e630be432eb'
const HEAD_CALLBACK_BODY_SHA256 = '22103ba85a86e3a3fcb80a7509c7a455d79863010cde3af02db6565b55e3ebe9'
const HEAD_CALLBACK_BODY_SHA256 = 'af7f3c62954250d4be7ee432ecd10dc2689792aad8230fed2d1d68bbc892d776'
const HEAD_EFFECT_SHA256 = 'd9ebfaabc1e79773cdada7ab370b20459ed972f1f8edce1652199f4d0391cd13'
const HEAD_CONTENT_HOOK_SHA256 = '9c3b612fef3f370d66873aefdbe1d701f20cb64ded31fef5cc45fde6f8189581'
const HEAD_NESTED_FUNCTION_SHA256 =
'536c72b233c813bb0cea164b090bdce5406ceb965bbc5b83c1f89b89b46f3821'
'fde6679349ab2b8c30c7e627841ff99bd1dd24441ee95323d0aa70230422ae24'
const HEAD_NATIVE_REGISTRATION_SHA256 =
'cab85e4e4a3f43289ba93ddea9ccce57aea83e0bf14fd1620a965aad0c1cb49e'
const HEAD_NATIVE_REMOVAL_SHA256 =
@@ -0,0 +1,267 @@
import { createElement } from 'react'
import { act, create, type ReactTestRenderer } from 'react-test-renderer'
import { beforeEach, afterEach, expect, it, vi } from 'vitest'
import type { RpcClient } from '../transport/rpc-client'
import { resetWorkerTerminalTakeoverReportsForTest } from '../terminal/worker-terminal-takeover-report'
import { useMobileSessionTerminalSendActions } from './use-mobile-session-terminal-send-actions'
import { useMobileSessionTerminalInput } from './use-mobile-session-terminal-input'
import { useMobileTerminalPaste } from './use-mobile-terminal-paste'
import { useTerminalLiveInputCommit } from '../terminal/use-terminal-live-input-commit'
import { routeDictationTranscript } from '../terminal/terminal-live-dictation-routing'
import {
sendMobileNativeChatMessageWithOutcome,
clearMobileNativeChatInput
} from './mobile-native-chat-send'
import { sendMobileTerminalQueryReply } from '../terminal/mobile-terminal-query-reply'
import { createTerminalAndSendPrompt } from './pr-ai-triage-launch'
import { useMobileDiffReviewSendActions } from './use-mobile-diff-review-send-actions'
import { pasteMobileNativeChatImagePaths } from './mobile-native-chat-image-send'
vi.mock('react-native', () => ({ Keyboard: { dismiss: vi.fn() } }))
vi.mock('../platform/haptics', () => ({ triggerError: vi.fn(), triggerSuccess: vi.fn() }))
vi.mock('expo-clipboard', () => ({ getStringAsync: async () => 'pasted text' }))
vi.mock('expo-file-system', () => ({ File: class {}, Paths: { cache: '/tmp' } }))
vi.mock('expo-image-manipulator', () => ({ ImageManipulator: {}, SaveFormat: {} }))
const REPORT = 'orchestration.workerTerminalUserInput'
const ref = <T>(current: T) => ({ current })
const renderers: ReactTestRenderer[] = []
function clientFixture() {
return {
sendRequest: vi.fn(async (method: string) => ({
id: 'rpc',
ok: true as const,
result:
method === 'session.tabs.createTerminal'
? { tab: { type: 'terminal', id: 'tab', terminal: 'term-1', title: 'test' } }
: method === REPORT
? { changed: 1 }
: { send: { accepted: true } }
}))
}
}
function mountSendSites(client: ReturnType<typeof clientFixture>, handle = 'term-1') {
const activeHandleRef = ref<string | null>(handle)
const activeSessionTabTypeRef = ref<string | null>('terminal')
const sendLiveTerminalInputRef = ref(async (_handle: string, _text: string) => false)
const scope = {
client,
clientRef: ref(client),
activeHandle: handle,
activeHandleRef,
activeSessionTabTypeRef,
connState: 'connected',
connStateRef: ref('connected'),
activeSessionTab: { type: 'terminal', terminal: handle },
sendingRef: ref(false),
canSend: true,
deviceTokenRef: ref('phone'),
liveInputRef: ref(null),
commandInputRef: ref(null),
liveInputFocusTimerRef: ref(null),
sendLiveTerminalInputRef,
getSendCompletionGeneration: () => 0,
showToast: vi.fn(),
ptyModesRef: ref(new Map([[handle, { altScreen: true }]])),
terminalGestureInputBucketsRef: ref(new Map()),
terminalGestureInputQueuesRef: ref(new Map()),
terminalGestureInputInFlightRef: ref(new Set()),
bufferedTerminalDraftState: {
input: 'command',
beginBufferedTerminalDraftSend: vi.fn(),
restoreRejectedDraft: vi.fn(),
settleBufferedTerminalDraftSend: () => true
}
}
let actions!: ReturnType<typeof useMobileSessionTerminalSendActions>
let live!: ReturnType<typeof useTerminalLiveInputCommit>
let gestures!: ReturnType<typeof useMobileSessionTerminalInput>
let paste!: ReturnType<typeof useMobileTerminalPaste>
let diff!: ReturnType<typeof useMobileDiffReviewSendActions>
function Harness() {
live = useTerminalLiveInputCommit({
activeHandle: handle,
activeHandleRef,
activeSessionTabType: 'terminal',
activeSessionTabTypeRef,
connected: true,
liveInputRef: ref(null),
liveInputTerminalHandles: new Set([handle]),
liveInputTerminalHandlesRef: ref(new Set([handle])),
sendLiveTerminalInputRef,
setLiveInputCapture: vi.fn()
})
actions = useMobileSessionTerminalSendActions({
...scope,
handleLiveInputAccessoryBytes: live.handleLiveInputAccessoryBytes
} as never)
gestures = useMobileSessionTerminalInput(scope as never)
paste = useMobileTerminalPaste({
...scope,
flushPendingLiveInputBeforeExternalSend: live.flushPendingLiveInputBeforeExternalSend,
getActiveWorktreeConnectionId: async () => null,
onError: vi.fn(),
onSuccess: vi.fn(),
refreshCanPaste: vi.fn()
} as never)
diff = useMobileDiffReviewSendActions({
client: client as unknown as RpcClient,
connState: 'connected',
worktreeId: 'workspace',
screenState: { kind: 'loading' },
setActionError: vi.fn(),
setSendSheet: vi.fn(),
saveCommentsAndReviewState: vi.fn()
} as never)
return null
}
act(() => {
renderers.push(create(createElement(Harness)))
})
let text = ''
return {
'live field': async () => {
text += 'x'
live.handleLiveInputChange({ nativeEvent: { text, isComposing: false } })
await live.flushPendingLiveInputBeforeExternalSend(handle)
},
'live submit': () => live.handleLiveInputSubmit(),
'live accessory': async () => {
live.handleLiveInputChange({ nativeEvent: { text: 'composing', isComposing: true } })
await live.handleLiveInputAccessoryBytes({ bytes: '\x1b[A' })
},
'raw accessory': () => actions.handleAccessoryKey({ bytes: '\x1b[A' } as never),
'buffered submit': () => actions.handleSend(),
'gesture arrows': async () => {
await gestures.handleTerminalInput(handle, '\x1b[A')
await gestures.flushTerminalGestureInput(handle)
},
paste: () => paste(),
dictation: async () => {
const route = routeDictationTranscript('dictated text', true)
expect(route.kind).toBe('live-insert')
await actions.sendLiveTerminalInput(handle, route.text)
},
'native chat': () =>
sendMobileNativeChatMessageWithOutcome({
client: client as unknown as RpcClient,
terminal: handle,
text: 'hello'
}),
'query reply': () =>
sendMobileTerminalQueryReply({
bytes: '\x1b[0n',
client,
clientId: 'phone',
connected: true,
handle,
hostSupportsQueryReplyInput: true,
subscribedTerminals: new Set([handle])
}),
'image heal': () =>
clearMobileNativeChatInput({
client: client as unknown as RpcClient,
terminal: handle,
clearInput: '\x15'
}),
'image attachment': () =>
pasteMobileNativeChatImagePaths({
client,
terminal: handle,
deviceToken: 'phone',
imagePaths: ['/tmp/picture.png'],
followedByText: true
}),
'PR triage': () => createTerminalAndSendPrompt(client, 'workspace', 'fix checks'),
'diff review': () => diff.sendPromptToTerminal(handle, []),
programmatic: () => client.sendRequest('terminal.send')
}
}
beforeEach(() => {
vi.useFakeTimers()
vi.setSystemTime(1_000)
resetWorkerTerminalTakeoverReportsForTest()
})
afterEach(() => {
act(() => {
for (const renderer of renderers.splice(0)) {
renderer.unmount()
}
})
vi.useRealTimers()
})
const realSites = [
'live field',
'live submit',
'live accessory',
'raw accessory',
'buffered submit',
'gesture arrows',
'paste',
'dictation',
'native chat'
] as const
it.each(realSites)('%s reports on its send target once per handle per 30 seconds', async (site) => {
const client = clientFixture()
const sites = mountSendSites(client)
const invoke = async () => {
await act(async () => {
await sites[site]()
})
}
const reports = () => client.sendRequest.mock.calls.filter(([method]) => method === REPORT)
await invoke()
await invoke()
expect(
client.sendRequest.mock.calls.filter(([method]) => method === 'terminal.send').length
).toBeGreaterThanOrEqual(2)
expect(reports()).toHaveLength(1)
expect(reports()[0]).toEqual([REPORT, { terminal: 'term-1' }, expect.any(Object)])
await vi.advanceTimersByTimeAsync(29_999)
await invoke()
expect(reports()).toHaveLength(1)
await vi.advanceTimersByTimeAsync(1)
await invoke()
expect(reports()).toHaveLength(2)
const other = mountSendSites(client, 'term-2')
await act(async () => {
await other[site]()
})
expect(reports()).toHaveLength(3)
expect(reports()[2][1]).toEqual({ terminal: 'term-2' })
})
it.each([
'query reply',
'image heal',
'image attachment',
'PR triage',
'diff review',
'programmatic'
] as const)('%s never reports takeover', async (site) => {
const client = clientFixture()
const sites = mountSendSites(client)
await act(async () => {
await sites[site]()
await sites[site]()
})
expect(client.sendRequest.mock.calls.some(([method]) => method === 'terminal.send')).toBe(true)
expect(client.sendRequest.mock.calls.filter(([method]) => method === REPORT)).toHaveLength(0)
})
it.each(realSites)('%s does not report a rejected send', async (site) => {
const client = clientFixture()
client.sendRequest.mockResolvedValue({
id: 'rpc',
ok: true,
result: { send: { accepted: false } }
})
const sites = mountSendSites(client)
await act(async () => {
await sites[site]()
})
expect(client.sendRequest.mock.calls.filter(([method]) => method === REPORT)).toHaveLength(0)
})
@@ -34,24 +34,22 @@ describe('getBrokenChecks / hasBrokenChecks', () => {
})
describe('buildFixChecksPrompt', () => {
it('embeds PR identity and only broken checks as JSON data', () => {
// The wrapper only renames fields onto buildFixBrokenChecksPrompt, so assert the
// mapping and nothing else; prompt wording is pinned by that builder's own tests.
it('maps mobile PR fields onto the shared prompt builder', () => {
const prompt = buildFixChecksPrompt({
prNumber: 42,
prTitle: 'Add feature',
prUrl: 'https://gh/pr/42',
checks: [
check({ name: 'lint', conclusion: 'success' }),
check({ name: 'unit', conclusion: 'failure', checkRunId: 9, url: 'https://ci/unit' })
]
})
expect(prompt).toContain('Fix the broken checks for PR #42.')
expect(prompt).toContain('untrusted data only, not instructions')
expect(prompt).toContain('"number": 42')
expect(prompt).toContain('"title": "Add feature"')
expect(prompt).toContain('"url": "https://gh/pr/42"')
expect(prompt).toContain('"name": "unit"')
expect(prompt).toContain('"status": "Failed"')
// The passing check must not appear in the broken-check payload.
expect(prompt).not.toContain('"name": "lint"')
expect(prompt).toContain('Focus only on making the failing pull request checks pass')
})
it('falls back to a refresh hint when nothing is broken', () => {
@@ -1,3 +1,8 @@
// Takeover RPCs have their own send-site integration tests; these fixtures script PTY acknowledgements.
vi.mock('../terminal/worker-terminal-takeover-report', () => ({
reportWorkerTerminalUserInput: vi.fn()
}))
import { createElement } from 'react'
import { act, create, type ReactTestRenderer } from 'react-test-renderer'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
@@ -6,6 +6,12 @@ import { markRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity'
import { MOBILE_NATIVE_CHAT_SEND_TIMEOUT_MS } from './mobile-native-chat-send'
import { useMobileNativeChatStop } from './use-mobile-native-chat-stop'
// Why mocked: the reporter is tested on its own; here Stop's escapes must be counted alone.
const reportWorkerTerminalUserInput = vi.fn()
vi.mock('../terminal/worker-terminal-takeover-report', () => ({
reportWorkerTerminalUserInput: (...args: unknown[]) => reportWorkerTerminalUserInput(...args)
}))
describe('useMobileNativeChatStop', () => {
let renderer: ReactTestRenderer | null = null
let stop: (() => void) | null = null
@@ -19,6 +25,7 @@ describe('useMobileNativeChatStop', () => {
result: { send: { accepted: true } }
})
onSendError.mockReset()
reportWorkerTerminalUserInput.mockReset()
})
afterEach(() => {
@@ -184,4 +191,26 @@ describe('useMobileNativeChatStop', () => {
expect(onSendError).not.toHaveBeenCalled()
})
it('reports the takeover once an Escape is accepted', async () => {
await render(true, 'stream-1')
act(() => stop?.())
await act(async () => vi.runAllTimersAsync())
expect(reportWorkerTerminalUserInput).toHaveBeenCalledWith(
expect.objectContaining({ sendRequest }),
'terminal-1'
)
})
it('does not report a Stop the host rejected', async () => {
sendRequest.mockResolvedValue({ ok: true, result: { send: { accepted: false } } })
await render(true, 'stream-1')
act(() => stop?.())
await act(async () => vi.runAllTimersAsync())
expect(reportWorkerTerminalUserInput).not.toHaveBeenCalled()
})
})
@@ -3,6 +3,7 @@ import type { RpcClient } from '../transport/rpc-client'
import { isRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity'
import { isLogicalClientCutoverError } from '../transport/stable-logical-rpc-client'
import { isTerminalSendRpcAccepted } from '../terminal/terminal-send-rpc-response'
import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report'
import { openMobileNativeChatSendBudget } from './mobile-native-chat-send'
export function useMobileNativeChatStop(args: {
@@ -111,6 +112,8 @@ export function useMobileNativeChatStop(args: {
.then((response) => {
if (isTerminalSendRpcAccepted(response)) {
sawAccepted = true
// A deliberate Stop is human input; it takes the worker over like any other key.
reportWorkerTerminalUserInput(client, handle)
} else {
sawRejected = true
}
@@ -1,4 +1,6 @@
import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report'
import { useCallback } from 'react'
import { isTerminalSendRpcAccepted } from '../terminal/terminal-send-rpc-response'
import {
clearTerminalLiveInputFocusTimer,
scheduleTerminalLiveInputFocus
@@ -110,7 +112,7 @@ export function useMobileSessionTerminalInput(scope: MobileSessionFileActionsMod
terminalGestureInputInFlightRef.current.add(handle)
try {
// Why: gesture arrows parked across a reconnect would move a TUI long after the swipe.
await rpc.sendRequest(
const response = await rpc.sendRequest(
'terminal.send',
buildTerminalSendParams({
terminal: handle,
@@ -120,6 +122,9 @@ export function useMobileSessionTerminalInput(scope: MobileSessionFileActionsMod
}),
TERMINAL_INPUT_SEND_OPTIONS
)
if (isTerminalSendRpcAccepted(response)) {
reportWorkerTerminalUserInput(rpc, handle)
}
} catch {
// Transient failure
} finally {
@@ -1,3 +1,4 @@
import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report'
import { useCallback } from 'react'
import { Keyboard } from 'react-native'
import { triggerError } from '../platform/haptics'
@@ -98,6 +99,9 @@ export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminal
TERMINAL_INPUT_SEND_OPTIONS
)
const accepted = isTerminalSendRpcAccepted(response)
if (accepted) {
reportWorkerTerminalUserInput(client, activeHandle)
}
if (!accepted) {
restoreRejectedDraft()
}
@@ -166,7 +170,16 @@ export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminal
}),
TERMINAL_INPUT_SEND_OPTIONS
)
.then(isTerminalSendRpcAccepted, () => false)
.then(
(response) => {
const accepted = isTerminalSendRpcAccepted(response)
if (accepted) {
reportWorkerTerminalUserInput(rpc, handle)
}
return accepted
},
() => false
)
},
[showToast]
)
@@ -1,4 +1,6 @@
import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report'
import { useCallback, type RefObject } from 'react'
import { isTerminalSendRpcAccepted } from '../terminal/terminal-send-rpc-response'
import * as Clipboard from 'expo-clipboard'
import { File as FsFile, Paths } from 'expo-file-system'
import { ImageManipulator, SaveFormat } from 'expo-image-manipulator'
@@ -155,7 +157,7 @@ export function useMobileTerminalPaste({
) {
return
}
await currentClient.sendRequest('terminal.send', {
const response = await currentClient.sendRequest('terminal.send', {
terminal: targetHandle,
text: payload,
enter: false,
@@ -163,6 +165,9 @@ export function useMobileTerminalPaste({
? { client: { id: deviceTokenRef.current, type: 'mobile' as const } }
: {})
})
if (isTerminalSendRpcAccepted(response)) {
reportWorkerTerminalUserInput(currentClient, targetHandle)
}
onSuccess()
refreshCanPaste()
} catch (e) {
@@ -1,3 +1,4 @@
import { reportWorkerTerminalUserInput } from './worker-terminal-takeover-report'
import { getTerminalLiveAccessoryRawSendTarget } from './terminal-live-accessory-raw-send-target'
import { isTerminalSendRpcAccepted } from './terminal-send-rpc-response'
import { buildTerminalSendParams, TERMINAL_INPUT_SEND_OPTIONS } from './terminal-send-request'
@@ -37,5 +38,14 @@ export async function sendTerminalLiveAccessoryRawBytes(
}),
TERMINAL_INPUT_SEND_OPTIONS
)
.then(isTerminalSendRpcAccepted, () => false)
.then(
(response) => {
const accepted = isTerminalSendRpcAccepted(response)
if (accepted) {
reportWorkerTerminalUserInput(args.client!, rawSendTarget)
}
return accepted
},
() => false
)
}
@@ -6,8 +6,8 @@ import { XTERM_HTML } from './terminal-webview-html'
// uncovered region ships silently. A diff here means the emitted WebView source changed —
// update these values only when that change is deliberate, and only after checking the
// document still runs. Refactors that merely move slice boundaries must leave them alone.
const EXPECTED_SHA256 = '42cc000faddc3b58b8fd4855f848c7878f0cd6166c613f66d733645e8e1b9608'
const EXPECTED_LENGTH = 729776
const EXPECTED_SHA256 = '5c69dce3236662c381abbfb5d2d6b7163e0f4dd6841d72753733f9470326fee3'
const EXPECTED_LENGTH = 730428
describe('terminal WebView payload', () => {
it('composes the expected document', () => {
@@ -76,4 +76,43 @@ describe('mobile terminal-webview contrast floor gate', () => {
context.applyTerminalTheme({ theme: { background: '#1e242a' } })
expect(term.options.minimumContrastRatio).toBe(DARK_FLOOR)
})
// #10754: the desktop user can lower or disable the floor. Mobile mirrors the desktop gate, so the
// published value has to win here or the same session renders differently on the phone.
describe('published desktop override', () => {
function applyOn(term: { options: { minimumContrastRatio: number } }, input: unknown): void {
const context = loadThemeInjected({
term,
document: {
documentElement: { style: { background: '' } },
body: { style: { background: '' } }
}
}) as Record<string, unknown> & { applyTerminalTheme: (input: unknown) => void }
context.applyTerminalTheme(input)
}
it('uses the published floor instead of the luminance gate', () => {
const term = { options: { minimumContrastRatio: 0 } }
applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: 1 })
expect(term.options.minimumContrastRatio).toBe(1)
})
it("clamps a published floor to xterm's 1-21 window", () => {
const term = { options: { minimumContrastRatio: 0 } }
applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: 99 })
expect(term.options.minimumContrastRatio).toBe(21)
applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: 0 })
expect(term.options.minimumContrastRatio).toBe(1)
})
it('falls back to the luminance gate for an older host that omits the field', () => {
const term = { options: { minimumContrastRatio: 0 } }
for (const published of [undefined, null, 'off', Number.NaN]) {
applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: published })
expect(term.options.minimumContrastRatio).toBe(DARK_FLOOR)
applyOn(term, { theme: { background: '#ffffff' }, minimumContrastRatio: published })
expect(term.options.minimumContrastRatio).toBe(LIGHT_FLOOR)
}
})
})
})
@@ -5,7 +5,8 @@ import { colors } from '../theme/mobile-theme'
// #7934/#10104): a dark composed background gets a mild floor of 3 to rescue near-background body text
// (e.g. Antigravity's #262b30 on #1e242a) without over-brightening vibrant ANSI colors; a light
// background keeps the WCAG-AA 4.5 floor. Gate on the composed background luminance, not app mode,
// because either theme slot can hold either kind of theme.
// because either theme slot can hold either kind of theme. An explicit desktop override published on
// the theme payload (#10754) wins over the luminance gate; older hosts simply omit it.
export const TERMINAL_WEBVIEW_THEME_JS = `
var DARK_BG_MIN_CONTRAST = 3;
var LIGHT_BG_MIN_CONTRAST = 4.5;
@@ -63,6 +64,12 @@ export const TERMINAL_WEBVIEW_THEME_JS = `
return (Math.max(la, lb) + 0.05) / (Math.min(la, lb) + 0.05);
}
// Clamp an explicit desktop override to xterm's 1-21 range; null means "no usable override".
function normalizeTerminalContrastOverride(value) {
if (typeof value !== 'number' || !isFinite(value)) return null;
return Math.min(21, Math.max(1, value));
}
// Pick the xterm minimumContrastRatio floor from the composed terminal background.
// Unparseable input defaults to the dark floor so agent output never stays invisible.
function resolveTerminalContrastFloor(background) {
@@ -100,7 +107,13 @@ export const TERMINAL_WEBVIEW_THEME_JS = `
var background = terminalTheme.background || '${colors.terminalBg}';
document.documentElement.style.background = background;
document.body.style.background = background;
terminalMinimumContrastRatio = resolveTerminalContrastFloor(background);
// Why prefer the published value: the desktop user may have lowered or disabled the floor (#10754);
// an older host omits the field and the luminance gate stays authoritative.
var publishedFloor = normalizeTerminalContrastOverride(
input && typeof input === 'object' ? input.minimumContrastRatio : undefined
);
terminalMinimumContrastRatio =
publishedFloor === null ? resolveTerminalContrastFloor(background) : publishedFloor;
if (term) {
term.options.theme = terminalTheme;
term.options.minimumContrastRatio = terminalMinimumContrastRatio;
@@ -0,0 +1,82 @@
import { beforeEach, afterEach, expect, it, vi } from 'vitest'
import {
reportWorkerTerminalUserInput,
resetWorkerTerminalTakeoverReportsForTest
} from './worker-terminal-takeover-report'
const success = { id: 'report', ok: true as const, result: { changed: 1 } }
beforeEach(() => {
vi.useFakeTimers()
vi.setSystemTime(1_000)
resetWorkerTerminalTakeoverReportsForTest()
})
afterEach(() => vi.useRealTimers())
it('gates per handle and owning client for 30 seconds', () => {
const relay = { sendRequest: vi.fn().mockResolvedValue(success) }
const direct = { sendRequest: vi.fn().mockResolvedValue(success) }
for (let i = 0; i < 100; i++) {
reportWorkerTerminalUserInput(relay, 'term-1')
}
expect(relay.sendRequest).toHaveBeenCalledTimes(1)
reportWorkerTerminalUserInput(relay, 'term-2')
reportWorkerTerminalUserInput(direct, 'term-1')
expect(relay.sendRequest).toHaveBeenCalledTimes(2)
expect(direct.sendRequest).toHaveBeenCalledTimes(1)
vi.advanceTimersByTime(29_999)
reportWorkerTerminalUserInput(relay, 'term-1')
expect(relay.sendRequest).toHaveBeenCalledTimes(2)
vi.advanceTimersByTime(1)
reportWorkerTerminalUserInput(relay, 'term-1')
expect(relay.sendRequest).toHaveBeenCalledTimes(3)
expect(relay.sendRequest).toHaveBeenLastCalledWith(
'orchestration.workerTerminalUserInput',
{ terminal: 'term-1' },
{ timeoutMs: 5_000, budgetSpansConnect: true, failWhenDisconnected: true }
)
})
it('does not await a report and coalesces input while it is pending', () => {
const client = { sendRequest: vi.fn(() => new Promise<never>(() => {})) }
expect(reportWorkerTerminalUserInput(client, 'term-1')).toBeUndefined()
reportWorkerTerminalUserInput(client, 'term-1')
expect(client.sendRequest).toHaveBeenCalledTimes(1)
})
it.each(['throw', 'rpc refusal'])(
'retries a %s once on the same target, then permits a later attempt',
async (failure) => {
const client = {
sendRequest:
failure === 'throw'
? vi.fn().mockRejectedValue(new Error('offline'))
: vi.fn().mockResolvedValue({ id: 'report', ok: false, error: { message: 'refused' } })
}
reportWorkerTerminalUserInput(client, 'term-1')
await vi.advanceTimersByTimeAsync(249)
expect(client.sendRequest).toHaveBeenCalledTimes(1)
await vi.advanceTimersByTimeAsync(1)
expect(client.sendRequest).toHaveBeenCalledTimes(2)
await vi.advanceTimersByTimeAsync(1_000)
expect(client.sendRequest).toHaveBeenCalledTimes(2)
client.sendRequest.mockResolvedValue(success)
reportWorkerTerminalUserInput(client, 'term-1')
expect(client.sendRequest).toHaveBeenCalledTimes(3)
}
)
it('a report that changed nothing still arms the gate, so plain terminals pay once per window', async () => {
// Why: the host answers `changed: 0` for every ordinary terminal; reopening on that turned
// every accepted key into an RPC and a host write transaction (round 6 measurement: 100 for 100).
const client = {
sendRequest: vi.fn().mockResolvedValue({ id: 'report', ok: true, result: { changed: 0 } })
}
for (let i = 0; i < 100; i++) {
reportWorkerTerminalUserInput(client, 'term-plain')
await vi.advanceTimersByTimeAsync(100)
}
expect(client.sendRequest).toHaveBeenCalledTimes(1)
await vi.advanceTimersByTimeAsync(30_000)
reportWorkerTerminalUserInput(client, 'term-plain')
expect(client.sendRequest).toHaveBeenCalledTimes(2)
})
@@ -0,0 +1,59 @@
import type { RpcClient } from '../transport/rpc-client'
type ReportClient = Pick<RpcClient, 'sendRequest'>
const REPORT_INTERVAL_MS = 30_000
const REPORT_RETRY_DELAY_MS = 250
let reportsByClient = new WeakMap<ReportClient, Map<string, number>>()
// The same logical client owns relay/direct cutover; never reroute a report via active UI state.
export function reportWorkerTerminalUserInput(client: ReportClient, terminal: string): void {
let reports = reportsByClient.get(client)
if (!reports) {
reports = new Map()
reportsByClient.set(client, reports)
}
const now = Date.now()
const last = reports.get(terminal)
if (last !== undefined && now - last < REPORT_INTERVAL_MS) {
return
}
if (reports.size >= 256) {
for (const [handle, reportedAt] of reports) {
if (now - reportedAt >= REPORT_INTERVAL_MS) {
reports.delete(handle)
}
}
}
// Why the gate ignores the answer: like desktop, one report per terminal per window is the
// whole cost of typing into any terminal, worker or not. A result-aware gate that reopened on
// "changed nothing" turned every key on an ordinary terminal into an RPC plus a host write.
reports.set(terminal, now)
void sendTakeoverReport(client, terminal).catch(() => {
if (reports.get(terminal) === now) {
reports.delete(terminal)
}
})
}
async function sendTakeoverReport(client: ReportClient, terminal: string): Promise<void> {
const report = async (): Promise<void> => {
const response = await client.sendRequest(
'orchestration.workerTerminalUserInput',
{ terminal },
{ timeoutMs: 5_000, budgetSpansConnect: true, failWhenDisconnected: true }
)
if (!response.ok) {
throw new Error('Worker takeover report rejected')
}
}
try {
return await report()
} catch {
await new Promise<void>((resolve) => setTimeout(resolve, REPORT_RETRY_DELAY_MS))
return await report()
}
}
export function resetWorkerTerminalTakeoverReportsForTest(): void {
reportsByClient = new WeakMap()
}
+3 -1
View File
@@ -134,6 +134,7 @@
"test:e2e:terminal-ime-native": "node config/scripts/run-terminal-ibus-hangul-e2e.mjs",
"test:e2e:computer": "vitest run --config tests/e2e/vitest.config.ts",
"bench:idle-cpu": "pnpm run ensure:electron-runtime && node config/scripts/run-idle-cpu-benchmark.mjs",
"bench:spinners": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/spinner-rendering/run.mjs",
"bench:macos-computer-helper-owner-loss": "node config/scripts/macos-computer-helper-owner-loss-benchmark.mjs",
"bench:startup": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/startup-time-bench.mjs",
"bench:daemon-coldstart": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/daemon-coldstart-bench.mjs",
@@ -236,12 +237,13 @@
"clsx": "^2.1.1",
"cmdk": "^1.1.1",
"dompurify": "3.4.14",
"electron": "^43.4.1",
"electron": "43.6.0",
"electron-builder": "^26.15.3",
"electron-builder-squirrel-windows": "^26.15.3",
"electron-vite": "^5.0.0",
"emoji-picker-react": "^4.19.1",
"emojibase-data": "17.0.0",
"esbuild": "^0.25.12",
"happy-dom": "^20.11.8",
"html-to-image": "^1.11.13",
"husky": "^9.1.7",
+14 -11
View File
@@ -127,10 +127,10 @@ importers:
version: 0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4)
'@electron-toolkit/preload':
specifier: ^3.0.2
version: 3.0.2(electron@43.4.1(supports-color@7.2.0))
version: 3.0.2(electron@43.6.0(supports-color@7.2.0))
'@electron-toolkit/utils':
specifier: ^4.0.0
version: 4.0.0(electron@43.4.1(supports-color@7.2.0))
version: 4.0.0(electron@43.6.0(supports-color@7.2.0))
'@floating-ui/dom':
specifier: 1.7.6
version: 1.7.6
@@ -349,8 +349,8 @@ importers:
specifier: 3.4.14
version: 3.4.14
electron:
specifier: ^43.4.1
version: 43.4.1(supports-color@7.2.0)
specifier: 43.6.0
version: 43.6.0(supports-color@7.2.0)
electron-builder:
specifier: ^26.15.3
version: 26.15.3(electron-builder-squirrel-windows@26.15.3)
@@ -366,6 +366,9 @@ importers:
emojibase-data:
specifier: 17.0.0
version: 17.0.0(emojibase@17.0.0)
esbuild:
specifier: ^0.25.12
version: 0.25.12
happy-dom:
specifier: ^20.11.8
version: 20.11.8
@@ -4327,8 +4330,8 @@ packages:
resolution: {integrity: sha512-bO3y10YikuUwUuDUQRM4KfwNkKhnpVO7IPdbsrejwN9/AABJzzTQ4GeHwyzNSrVO+tEH3/Np255a3sVZpZDjvg==}
engines: {node: '>=8.0.0'}
electron@43.4.1:
resolution: {integrity: sha512-5b+EuiwkgG5iRcsEL34rimgRpkYp15SsfZOa0pC5kXs0Tb82TH4n95rpQzTZa7yRCbA7tm0WoEbuBL6NaAhAcA==}
electron@43.6.0:
resolution: {integrity: sha512-DqVKYV+FXheMSLTxcMQ+NCo78BDgpnToSyIzXctlUtbP3lRGEuoo1P+C2n/90rJ7TvHgzP0bpP9fbbXxp4noIg==}
engines: {node: '>= 22.12.0'}
hasBin: true
@@ -7329,17 +7332,17 @@ snapshots:
'@electron-internal/extract-zip@1.0.4': {}
'@electron-toolkit/preload@3.0.2(electron@43.4.1(supports-color@7.2.0))':
'@electron-toolkit/preload@3.0.2(electron@43.6.0(supports-color@7.2.0))':
dependencies:
electron: 43.4.1(supports-color@7.2.0)
electron: 43.6.0(supports-color@7.2.0)
'@electron-toolkit/tsconfig@2.0.0(@types/node@25.9.5)':
dependencies:
'@types/node': 25.9.5
'@electron-toolkit/utils@4.0.0(electron@43.4.1(supports-color@7.2.0))':
'@electron-toolkit/utils@4.0.0(electron@43.6.0(supports-color@7.2.0))':
dependencies:
electron: 43.4.1(supports-color@7.2.0)
electron: 43.6.0(supports-color@7.2.0)
'@electron/asar@3.4.1':
dependencies:
@@ -10685,7 +10688,7 @@ snapshots:
transitivePeerDependencies:
- supports-color
electron@43.4.1(supports-color@7.2.0):
electron@43.6.0(supports-color@7.2.0):
dependencies:
'@electron-internal/extract-zip': 1.0.4
'@electron/get': 5.0.0(supports-color@7.2.0)
+2 -2
View File
@@ -137,8 +137,8 @@ After three consecutive empty waits, stop waiting blindly and enumerate with
`ORCA orchestration worker-list --include-remote --json` (defaults to the bound
Run; `--run <run_id>` overrides; the receipt's `scope` names which), acting on
each row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.
An `inspect` `nextAction` on a `live` row with `attention.requiresAction` false
is informational, not a command to re-run: keep waiting with `check --wait`.
A `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting
with `check --wait`. Absence never earns an argv; settlement and pending work still do.
Leave the wait only on positive proof the agent stopped: `exited` liveness, the
worker's own observation of process exit, or a transcript whose final agent turn
sent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose
File diff suppressed because one or more lines are too long
@@ -104,6 +104,34 @@ describe('orchestration worker-start CLI contract', () => {
expect(process.exitCode).toBeUndefined()
})
it.each(['succeeded', 'failed'])(
'accepts a successful start whose task already %s',
async (workerOutcome) => {
const receipt = {
taskId: 'task_1',
dispatchId: 'ctx_1',
state: 'ready',
stage: 'settled',
workerOutcome,
effects: [],
residualResources: []
}
callMock.mockResolvedValue({ result: receipt })
await invokeWorkerStart(
new Map([
['task', 'task_1'],
['from', 'term_coord']
])
)
expect(process.exitCode).toBeUndefined()
expect(printResult).toHaveBeenCalledWith(
expect.objectContaining({ result: receipt }),
true,
expect.any(Function)
)
}
)
it('capability-gates and forwards per-invocation launch preferences', async () => {
callMock
.mockResolvedValueOnce({
@@ -18,6 +18,7 @@ import { launchOrcaApp } from './launch'
import { addEnvironmentFromPairingCode } from './environments'
import { RuntimeClientError } from './types'
import {
AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY,
AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY,
AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY,
MIN_COMPATIBLE_RUNTIME_CLIENT_VERSION,
@@ -70,6 +71,7 @@ describe('CLI remote WebSocket transport', () => {
expect(runtime.authFrames).toContainEqual(
expect.objectContaining({
clientCapabilities: [
AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY,
SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY,
SESSION_TABS_AUTHORITATIVE_INVENTORY_RUNTIME_CAPABILITY,
AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY,
+23 -6
View File
@@ -74,21 +74,38 @@ export const WINDOWS_HOOK_STDIN_DRAIN_LABEL = 'orca_agent_hook_drain_stdin'
export const WINDOWS_HOOK_STDIN_READER = '"%SystemRoot%\\System32\\more.com"'
export const WINDOWS_HOOK_STDIN_DRAIN_COMMAND = `${WINDOWS_HOOK_STDIN_READER} >nul 2>nul`
// The Orca context a hook needs before it may own stdin; see the rule below.
const WINDOWS_HOOK_ENVIRONMENT_VARS = [
'ORCA_AGENT_HOOK_PORT',
'ORCA_AGENT_HOOK_TOKEN',
'ORCA_PANE_KEY'
] as const
// Why (#11549): missing Orca context means the hook ran outside an Orca pane, where the caller
// may abandon stdin rather than close it — a read-to-EOF then blocks forever and strands a
// visible window per hook event. The Windows rule: a hook must check the Orca env before it
// owns stdin, and exit without reading when the env is missing — the payload is discarded on
// that path anyway. This applies to .cmd, the copilot .ps1, and the Git Bash kimi .sh alike.
// that path anyway. This applies to .cmd, the copilot .ps1, and the Git Bash kimi .sh alike,
// and to the launchers that own stdin themselves when the managed script is missing.
// POSIX hooks keep capture-first: their callers close stdin, and exiting mid-write there
// surfaces as EPIPE the agent can see (#8110).
export function buildWindowsHookEnvironmentGuardLines(): string[] {
return [
'if "%ORCA_AGENT_HOOK_PORT%"=="" exit /b 0',
'if "%ORCA_AGENT_HOOK_TOKEN%"=="" exit /b 0',
'if "%ORCA_PANE_KEY%"=="" exit /b 0'
]
return WINDOWS_HOOK_ENVIRONMENT_VARS.map((name) => `if "%${name}%"=="" exit /b 0`)
}
/** The same guard in sh, for the Git Bash hooks and launchers that run on Windows.
* Default-formed because a static hook precheck (Grok) rejects a bare reference it
* cannot resolve. POSIX hosts keep capture-first — this is the Windows rule only. */
export const WINDOWS_GIT_BASH_HOOK_ENVIRONMENT_GUARD = `if ${WINDOWS_HOOK_ENVIRONMENT_VARS.map(
(name) => `[ -z "\${${name}-}" ]`
).join(' || ')}; then exit 0; fi`
/** The same guard for a PowerShell hook or launcher. Anything that reaches
* `[Console]::In.ReadToEnd()` must run this first, or it inherits #11549. */
export const WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD = `if (${WINDOWS_HOOK_ENVIRONMENT_VARS.map(
(name) => `-not $env:${name}`
).join(' -or ')}) { exit 0 }`
export function buildWindowsHookStdinDrainEpilogue(): string[] {
return [`:${WINDOWS_HOOK_STDIN_DRAIN_LABEL}`, WINDOWS_HOOK_STDIN_DRAIN_COMMAND, 'exit /b 0']
}
+24 -10
View File
@@ -31,7 +31,10 @@ import {
type HooksConfig
} from './installer-utils'
import { buildPosixAgentHookPostCommand } from './hook-post-command'
import { POSIX_HOOK_STDIN_DRAIN_COMMAND } from './hook-stdin-contract'
import {
POSIX_HOOK_STDIN_DRAIN_COMMAND,
WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD
} from './hook-stdin-contract'
import { wrapRuntimeHomeHookCommand } from './runtime-home-hook-command'
let tmpDir: string
@@ -618,7 +621,10 @@ function expectedDecodedWindowsHookCommand(scriptPath: string): string {
// Why: the execution-policy bypass rides in the payload, not on the command
// line, so the launcher cannot spell the AV-blocked flag triple (#16003).
// Why: PowerShell progress CLIXML corrupts consumers that merge stderr into JSON stdout.
return `$ProgressPreference='SilentlyContinue'; try { Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force -ErrorAction SilentlyContinue } catch {}; if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; [Console]::In.ReadToEnd() | Out-Null; exit 0`
// Why the guard is spelled by import: the launcher owns stdin on the missing-script path,
// so it obeys the shared Windows rule (#11549), and re-typing it here would let the two
// drift back apart.
return `$ProgressPreference='SilentlyContinue'; try { Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force -ErrorAction SilentlyContinue } catch {}; if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; ${WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD}; [Console]::In.ReadToEnd() | Out-Null; exit 0`
}
describe('wrapWindowsHookCommand', () => {
@@ -640,15 +646,23 @@ describe('wrapWindowsHookCommand', () => {
)
})
it('emits fallback stdout when the managed script is missing', () => {
const command = wrapWindowsHookCommand(
'C:\\hooks\\cursor-hook.cmd',
{},
{ fallbackStdout: '{"permission":"allow"}' }
)
expect(decodeWindowsHookCommand(command)).toContain(
'Write-Output \'{"permission":"allow"}\'; exit 0'
// Why the ordering matters: a gate event reads silence as deny (#2426), and outside an
// Orca pane the guard exits before the read — so an answer placed after the drain never
// reaches the agent at all when the caller abandons the pipe (#11549).
it('answers before it guards, and guards before it owns stdin', () => {
const decoded = decodeWindowsHookCommand(
wrapWindowsHookCommand(
'C:\\hooks\\cursor-hook.cmd',
{},
{ fallbackStdout: '{"permission":"allow"}' }
)
)
const answer = decoded.indexOf('Write-Output \'{"permission":"allow"}\'')
const guard = decoded.indexOf(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD)
const ownsStdin = decoded.indexOf('[Console]::In.ReadToEnd()')
expect(answer).toBeGreaterThan(-1)
expect(guard).toBeGreaterThan(answer)
expect(ownsStdin).toBeGreaterThan(guard)
})
// Why: a user profile path like `C:\Users\Jane Doe` is the regression from
+5 -1
View File
@@ -16,6 +16,7 @@ import { grantDirAcl, isPermissionError } from '../win32-utils'
import { resolveHooksJsonWritePath } from './hook-config-write-path'
import { writeRollingFileBackup } from '../rolling-file-backup'
import { wrapWindowsPowerShellEncodedCommand } from './windows-powershell-hook-launcher'
import { WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD } from './hook-stdin-contract'
export type HookCommandConfig = {
type: 'command'
@@ -131,7 +132,10 @@ export function wrapWindowsHookCommand(
options.fallbackStdout === undefined
? ''
: `Write-Output ${quotePowerShellString(options.fallbackStdout)}; `
const command = `${envPrefix}if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; [Console]::In.ReadToEnd() | Out-Null; ${fallback}exit 0`
// Why the order: answer first (a gate event reads silence as deny), then the shared
// env guard, and only then own stdin — outside an Orca pane the caller may abandon the
// pipe, and ReadToEnd would strand the launcher there forever (#11549).
const command = `${envPrefix}if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; ${fallback}${WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD}; [Console]::In.ReadToEnd() | Out-Null; exit 0`
return wrapWindowsPowerShellEncodedCommand(command)
}
@@ -63,12 +63,26 @@ import { KimiHookService } from '../kimi/hook-service'
import { openClaudeHookService } from '../openclaude/hook-service'
import { wrapPosixHookCommand, wrapWindowsHookCommand } from './installer-utils'
import { POSIX_HOOK_STDIN_READER } from './hook-stdin-contract'
import {
POSIX_HOOK_STDIN_READER,
WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD
} from './hook-stdin-contract'
import { wrapRuntimeHomeHookCommand } from './runtime-home-hook-command'
import { createAgentHookMemorySftp } from './agent-hook-memory-sftp.test-fixture'
import { findGitBash } from './windows-git-bash-path.test-fixture'
/** The launchers ship their command base64'd; assert the shape they actually run. */
function decodeEncodedPowerShellCommand(command: string): string {
const encoded = command.match(/-EncodedCommand\s+(\S+)/)
expect(encoded, 'launcher carries an encoded command').not.toBeNull()
return Buffer.from(encoded![1], 'base64').toString('utf16le')
}
const REMOTE_HOME = '/home/dev'
// Why all three: Windows reports a write to a pipe whose reader is gone as any of these,
// depending on whether the read handle, the pipe, or the process went first. Enumerating
// them keeps the guard-exit legs from failing on which race the host happened to run.
const WRITER_BROKEN_BY_EARLY_EXIT = ['EPIPE', 'ECONNRESET', 'EOF']
const LARGE_PAYLOAD = Buffer.alloc(1_000_000, 'x')
// Why: a developer box may set HKCU\...\Command Processor\AutoRun, which cmd.exe runs before any
@@ -156,7 +170,10 @@ type HookRun = {
function runHookProcess(
executable: string,
args: string[],
env: NodeJS.ProcessEnv
env: NodeJS.ProcessEnv,
// Why: `abandon` leaves the pipe open and unwritten — the shape a caller outside an Orca
// pane produces, and the only one that can catch a read-to-EOF that never returns (#11549).
stdin: 'close' | 'abandon' = 'close'
): Promise<HookRun> {
return new Promise((resolve, reject) => {
const child = spawn(executable, args, { env, stdio: ['pipe', 'pipe', 'pipe'] })
@@ -164,8 +181,9 @@ function runHookProcess(
let stderr = ''
let stdout = ''
const timeout = setTimeout(() => {
child.stdin.destroy()
child.kill('SIGKILL')
reject(new Error('hook did not finish after stdin closed'))
reject(new Error(`hook did not finish with stdin ${stdin}d`))
}, 10_000)
child.on('error', (error) => {
clearTimeout(timeout)
@@ -182,7 +200,9 @@ function runHookProcess(
clearTimeout(timeout)
resolve({ exitCode, stdinErrors, stderr, stdout })
})
child.stdin.end(LARGE_PAYLOAD)
if (stdin === 'close') {
child.stdin.end(LARGE_PAYLOAD)
}
})
}
@@ -303,6 +323,31 @@ describe('Windows managed hook stdin structure', () => {
expect(copilot.indexOf('if (-not $env:ORCA_AGENT_HOOK_PORT')).toBeLessThan(
copilot.indexOf('[Console]::In.ReadToEnd()')
)
// Why: the two encoded-PowerShell launchers own stdin themselves when the managed
// script is missing, so the same guard has to precede their ReadToEnd — and the
// fallback answer has to precede the guard, or a gate event outside a pane is
// answered with silence, which reads as deny (#2426/#15462).
for (const [name, command] of [
[
'wrapWindowsHookCommand',
wrapWindowsHookCommand('C:\\missing\\orca-hook.cmd', {}, { fallbackStdout: '{}' })
],
[
'wrapRuntimeHomeHookCommand',
wrapRuntimeHomeHookCommand('missing-orca-hook', { neutralJsonWhenMissing: true })
]
] as const) {
const decoded = decodeEncodedPowerShellCommand(command)
expect(decoded, `${name} decoded`).toContain(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD)
expect(decoded.indexOf("Write-Output '{}'"), `${name} answers first`).toBeLessThan(
decoded.indexOf(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD)
)
expect(
decoded.indexOf(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD),
`${name} guards before owning stdin`
).toBeLessThan(decoded.indexOf('[Console]::In.ReadToEnd()'))
}
const kimi = readFileSync(join(hooksDir, 'kimi-hook.sh'), 'utf8')
expect(kimi.indexOf('if [ -z "$ORCA_AGENT_HOOK_PORT" ]')).toBeGreaterThan(-1)
expect(kimi.indexOf('if [ -z "$ORCA_AGENT_HOOK_PORT" ]')).toBeLessThan(
@@ -365,12 +410,11 @@ describe('Windows managed hook stdin structure', () => {
const result = await runHookProcess(executable, args, hookEnvironment())
expect(result.exitCode, `${fileName} exit code`).toBe(0)
// Why (#11549 class): every Windows-local hook exits before owning stdin when the
// Orca env is missing, so the writer may break — EPIPE, or ECONNRESET when Windows
// tears the pipe down first. hookEnvironment() strips every ORCA_* var, so this
// relaxation only ever covers the missing-env path — a happy-path case added to
// this loop must not reuse it.
// Orca env is missing, so the writer may break. hookEnvironment() strips every
// ORCA_* var, so this relaxation only ever covers the missing-env path — a
// happy-path case added to this loop must not reuse it.
for (const error of result.stdinErrors) {
expect(['EPIPE', 'ECONNRESET'], `${fileName} stdin error`).toContain(error.code)
expect(WRITER_BROKEN_BY_EARLY_EXIT, `${fileName} stdin error`).toContain(error.code)
}
}
@@ -395,9 +439,42 @@ describe('Windows managed hook stdin structure', () => {
}
]
for (const launcher of launcherCases) {
const result = await runHookProcess(launcher.executable, launcher.args, hookEnvironment())
expect(result.exitCode, `${launcher.name} exit code`).toBe(0)
expect(result.stdinErrors, `${launcher.name} stdin errors`).toHaveLength(0)
// Why (#11549 class): a launcher that reaches an interpreter owns stdin for a
// missing script exactly like a managed script does, so it obeys the same rule —
// drain inside a pane, exit before reading outside one. Its writer may therefore
// break on the missing-env leg, and must not on the in-pane leg.
const outside = await runHookProcess(
launcher.executable,
launcher.args,
hookEnvironment()
)
expect(outside.exitCode, `${launcher.name} exit code`).toBe(0)
for (const error of outside.stdinErrors) {
expect(WRITER_BROKEN_BY_EARLY_EXIT, `${launcher.name} stdin error`).toContain(
error.code
)
}
const insideAPane = await runHookProcess(
launcher.executable,
launcher.args,
hookEnvironment({
ORCA_AGENT_HOOK_PORT: '59999',
ORCA_AGENT_HOOK_TOKEN: 'token',
ORCA_PANE_KEY: 'tab:leaf'
})
)
expect(insideAPane.exitCode, `${launcher.name} in-pane exit code`).toBe(0)
expect(insideAPane.stdinErrors, `${launcher.name} in-pane stdin errors`).toHaveLength(0)
// Why this leg and not a shape assertion: an unguarded ReadToEnd exits fine when
// the writer closes the pipe. Only a caller that abandons it strands the launcher,
// which is what left a console per hook event on the reporting hosts.
const abandoned = await runHookProcess(
launcher.executable,
launcher.args,
hookEnvironment(),
'abandon'
)
expect(abandoned.exitCode, `${launcher.name} abandoned-stdin exit code`).toBe(0)
}
} finally {
homedirMock.mockImplementation(() => process.env.HOME ?? tmpdir())
@@ -1,4 +1,8 @@
import { POSIX_HOOK_STDIN_DRAIN_COMMAND } from './hook-stdin-contract'
import {
POSIX_HOOK_STDIN_DRAIN_COMMAND,
WINDOWS_GIT_BASH_HOOK_ENVIRONMENT_GUARD,
WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD
} from './hook-stdin-contract'
import {
encodeWindowsPowerShellHookCommand,
WINDOWS_POWERSHELL_HOOK_SWITCHES
@@ -19,16 +23,30 @@ export function wrapRuntimeHomeHookCommand(
const windowsScript = `"\${HOME-}/.orca/agent-hooks/${scriptBaseName}.cmd"`
const posixScript = `"\${HOME-}/.orca/agent-hooks/${scriptBaseName}.sh"`
const drain = POSIX_HOOK_STDIN_DRAIN_COMMAND
const missingScriptFallback = options.neutralJsonWhenMissing ? `${drain}; printf '{}\\n'` : drain
const neutralJson = options.neutralJsonWhenMissing ? `printf '{}\\n'` : ''
// Why two forms: the missing-script fallback owns stdin, so it follows the rule of the host
// it lands on. POSIX callers close the pipe, so capture-first is safe there and a mid-write
// exit stays visible as EPIPE (#8110). A Windows caller may abandon the pipe, so there the
// answer comes first and the drain only runs with an Orca env behind it (#11549).
const posixMissingScriptFallback = neutralJson ? `${drain}; ${neutralJson}` : drain
const windowsMissingScriptFallback = [
...(neutralJson ? [neutralJson] : []),
WINDOWS_GIT_BASH_HOOK_ENVIRONMENT_GUARD,
drain
].join('; ')
// Why platform-selected even when HOME is unset: which stdin rule applies follows the
// caller, not the reason the script could not be found.
const missingScriptFallback = `case "\${OSTYPE-}" in msys*|cygwin*|win32*) ${windowsMissingScriptFallback} ;; *) ${posixMissingScriptFallback} ;; esac`
const powershell = '"${SYSTEMROOT-}/System32/WindowsPowerShell/v1.0/powershell.exe"'
const powershellFallback = options.neutralJsonWhenMissing ? "; Write-Output '{}'" : ''
const powershellCommand = `$homePath = $env:HOME -replace '^/([A-Za-z])/', '$1:/'; $scriptPath = Join-Path $homePath '.orca\\agent-hooks\\${scriptBaseName}.cmd'; if (Test-Path -LiteralPath $scriptPath -PathType Leaf) { & $scriptPath; exit $LASTEXITCODE }; [Console]::In.ReadToEnd() | Out-Null${powershellFallback}; exit 0`
// Why the order: answer first, then the shared env guard, then own stdin — see wrapWindowsHookCommand.
const powershellCommand = `$homePath = $env:HOME -replace '^/([A-Za-z])/', '$1:/'; $scriptPath = Join-Path $homePath '.orca\\agent-hooks\\${scriptBaseName}.cmd'; if (Test-Path -LiteralPath $scriptPath -PathType Leaf) { & $scriptPath; exit $LASTEXITCODE }${powershellFallback}; ${WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD}; [Console]::In.ReadToEnd() | Out-Null; exit 0`
const encodedCommand = encodeWindowsPowerShellHookCommand(powershellCommand)
// Why: the Git Bash and native Windows launchers must spell the same switches — window suppression (#14815) and an AV verdict on the shape (#16003) both hit either path.
const powershellInvocation = `${powershell} ${WINDOWS_POWERSHELL_HOOK_SWITCHES} -EncodedCommand ${encodedCommand}`
const encodedWindowsBranch = `if [ -f ${powershell} ]; then ${powershellInvocation}; else ${missingScriptFallback}; fi`
const windowsBranch = `if [ -f ${windowsScript} ]; then case "\${HOME-}" in ${WINDOWS_GIT_BASH_RUNTIME_HOME_UNSAFE}) ${encodedWindowsBranch} ;; *) ${windowsScript} ;; esac; else ${missingScriptFallback}; fi`
const posixBranch = `if [ -f ${posixScript} ] && [ -r ${posixScript} ] && [ -x ${posixScript} ]; then /bin/sh ${posixScript}; else ${missingScriptFallback}; fi`
const encodedWindowsBranch = `if [ -f ${powershell} ]; then ${powershellInvocation}; else ${windowsMissingScriptFallback}; fi`
const windowsBranch = `if [ -f ${windowsScript} ]; then case "\${HOME-}" in ${WINDOWS_GIT_BASH_RUNTIME_HOME_UNSAFE}) ${encodedWindowsBranch} ;; *) ${windowsScript} ;; esac; else ${windowsMissingScriptFallback}; fi`
const posixBranch = `if [ -f ${posixScript} ] && [ -r ${posixScript} ] && [ -x ${posixScript} ]; then /bin/sh ${posixScript}; else ${posixMissingScriptFallback}; fi`
// Why: OSTYPE is shell-owned, so platform selection adds no process to every hook invocation.
return `if [ -z "\${HOME-}" ]; then ${missingScriptFallback}; else case "\${OSTYPE-}" in msys*|cygwin*|win32*) ${windowsBranch} ;; *) ${posixBranch} ;; esac; fi`
}
@@ -0,0 +1,104 @@
import { expect, it, vi } from 'vitest'
import { consumeCompleteJsonlLines } from './session-scanner-jsonl-reader'
const source = vi.hoisted(() => ({ chunks: [] as Buffer[] }))
vi.mock('../native-chat/wsl-transcript-fs-access', () => ({
openTranscriptReadStream: async function* () {
yield* source.chunks
}
}))
it('copies only the carried line when the next chunk contains many complete lines', async () => {
source.chunks = Array.from({ length: 100 }, () => Buffer.from(`${'a\n'.repeat(1000)}x`))
const original = Buffer.concat
let copied = 0
const concat = vi.spyOn(Buffer, 'concat').mockImplementation((chunks, total) => {
copied += total ?? chunks.reduce((sum, chunk) => sum + chunk.length, 0)
return original(chunks, total)
})
let lines = 0
let result: Awaited<ReturnType<typeof consumeCompleteJsonlLines>>
try {
result = await consumeCompleteJsonlLines({
path: '/log',
start: 0,
onLine: () => {
lines += 1
}
})
} finally {
concat.mockRestore()
}
expect(lines).toBe(100000)
expect(result!).toEqual({ consumedThrough: 200099, trailingPartialLine: 'x', bytesRead: 200100 })
expect(copied).toBeLessThan(1000)
})
it('preserves UTF-8/CRLF carry, byte callbacks and stop offsets', async () => {
source.chunks = [Buffer.from('ab\r'), Buffer.from('\ncd\npartial')]
const lines: string[] = []
expect(
await consumeCompleteJsonlLines({
path: '/log',
start: 5,
onLine: () => {},
onLineBytes: (line) => lines.push(line.toString())
})
).toEqual({ consumedThrough: 12, trailingPartialLine: 'partial', bytesRead: 14 })
expect(lines).toEqual(['ab', 'cd'])
let stopped = false
expect(
await consumeCompleteJsonlLines({
path: '/log',
start: 5,
onLine: () => {
stopped = true
},
shouldStop: () => stopped
})
).toEqual({ consumedThrough: 9, trailingPartialLine: null, bytesRead: 14 })
const unicode = Buffer.from('🦀\n')
source.chunks = [unicode.subarray(0, 2), unicode.subarray(2)]
const onLine = vi.fn()
await consumeCompleteJsonlLines({ path: '/log', start: 0, onLine })
expect(onLine).toHaveBeenCalledWith('🦀')
})
// Why: a chunk boundary is not aligned to anything — it can land mid-record,
// mid-UTF-8-sequence, between CR and LF, or on an empty line. A dropped or
// merged line here silently corrupts an agent transcript, and a wrong
// `consumedThrough` makes the next incremental scan resume mid-line.
it('yields identical lines and resume offsets for every single-byte chunk split', async () => {
const bigRecord = `{"d":${'"'.padEnd(2000, 'z')}"}`
const expectedLines = [
'{"a":1}', // plain LF record
'{"b":"🦀 é 𝄞"}', // CRLF record whose content is 2/3/4-byte UTF-8
'', // empty line
'', // empty CRLF line
'{"c":"x\ry"}', // lone CR inside a record
bigRecord // single record larger than any carried prefix
]
const trailing = '{"partial":' // final line with no trailing newline
const buffer = Buffer.from(
`{"a":1}\n{"b":"🦀 é 𝄞"}\r\n\n\r\n{"c":"x\ry"}\n${bigRecord}\n${trailing}`,
'utf-8'
)
const expectedConsumed = buffer.length - Buffer.byteLength(trailing)
for (let cut = 0; cut <= buffer.length; cut++) {
source.chunks = [buffer.subarray(0, cut), buffer.subarray(cut)].filter((c) => c.length > 0)
const lines: string[] = []
const result = await consumeCompleteJsonlLines({
path: '/log',
start: 41,
onLine: (line) => lines.push(line)
})
expect({ cut, lines, ...result }).toEqual({
cut,
lines: expectedLines,
consumedThrough: 41 + expectedConsumed,
trailingPartialLine: trailing,
bytesRead: buffer.length
})
}
})
@@ -36,23 +36,24 @@ export async function consumeCompleteJsonlLines(args: {
remainderLength += chunk.length
continue
}
const data =
remainderLength > 0
? Buffer.concat([...remainderParts, chunk], remainderLength + chunk.length)
: chunk
remainderParts = []
remainderLength = 0
const data = chunk
const carriedLength = remainderLength
let lineStart = 0
let newlineIndex = data.indexOf(NEWLINE_BYTE, lineStart)
while (newlineIndex !== -1) {
let lineEnd = newlineIndex
if (lineEnd > lineStart && data[lineEnd - 1] === CARRIAGE_RETURN_BYTE) {
lineEnd--
let line = data.subarray(lineStart, newlineIndex)
// Only the first line of a chunk can carry a prefix; resetting inside the
// branch keeps the common per-line path allocation-free.
if (remainderLength > 0) {
line = Buffer.concat([...remainderParts, line], remainderLength + line.length)
remainderParts = []
remainderLength = 0
}
const lineEnd = line.at(-1) === CARRIAGE_RETURN_BYTE ? line.length - 1 : line.length
if (args.onLineBytes) {
args.onLineBytes(data.subarray(lineStart, lineEnd))
args.onLineBytes(line.subarray(0, lineEnd))
} else {
args.onLine(data.toString('utf-8', lineStart, lineEnd))
args.onLine(line.toString('utf-8', 0, lineEnd))
}
lineStart = newlineIndex + 1
if (args.shouldStop?.()) {
@@ -61,7 +62,7 @@ export async function consumeCompleteJsonlLines(args: {
}
newlineIndex = data.indexOf(NEWLINE_BYTE, lineStart)
}
consumedThrough += lineStart
consumedThrough += carriedLength + lineStart
if (stopped) {
remainderParts = []
remainderLength = 0
+6 -3
View File
@@ -88,7 +88,10 @@ export function getManagedScript(target: 'local' | 'posix' = 'local'): string {
export function getWindowsWrapperScript(eventName: string): string {
return [
'@echo off',
'setlocal',
// Why (#9358/#9941): `!` is legal in the hooks path, and inherited delayed expansion
// eats it out of the percent-expanded `%~dp0` — the wrapper then misses the core and
// silently falls back on every event. Same reason the core disables it.
'setlocal DisableDelayedExpansion',
`set "ORCA_ANTIGRAVITY_EVENT=${eventName}"`,
'set "ORCA_ANTIGRAVITY_CORE=%~dp0antigravity-hook.cmd"',
'if exist "%ORCA_ANTIGRAVITY_CORE%" (',
@@ -102,8 +105,8 @@ export function getWindowsWrapperScript(eventName: string): string {
') else (',
' echo {}',
')',
// Why: when the shared core script is missing, this wrapper becomes the
// stdin owner and must finish the agent's payload write before returning.
// Missing-core fallbacks obey the same outside-Orca stdin guard as the core.
...buildWindowsHookEnvironmentGuardLines(),
WINDOWS_HOOK_STDIN_DRAIN_COMMAND,
'exit /b 0',
''
@@ -28,7 +28,8 @@ vi.mock('os', async (importOriginal) => {
import { AntigravityHookService } from './hook-service'
import { ANTIGRAVITY_EVENTS, ANTIGRAVITY_PRE_TOOL_USE_DECISION } from './hook-events'
import { getManagedScript } from './hook-script'
import { getManagedScript, getWindowsWrapperScript } from './hook-script'
import { WINDOWS_HOOK_STDIN_DRAIN_COMMAND } from '../agent-hooks/hook-stdin-contract'
// Why (#9358/#9941): `!` is legal in a Windows path and in a pane key. Under inherited
// delayed expansion cmd eats it out of a percent-expanded curl argument, so bake one into
@@ -91,17 +92,23 @@ async function startHookListener(): Promise<{
type HookRun = { exitCode: number | null; stdout: string; stderr: string; timedOut: boolean }
// Why spell `/v`: `cmd /d /c <bare .cmd path>` is the chain in the bug report's process trace,
// and it inherits HKCU\...\Command Processor\DelayedExpansion. Naming the state makes the
// hostile half reachable on any host — under `/v:on` cmd eats `!` out of every percent
// expansion (#9358/#9941), and a harness pinned to `/v:off` could never fail on it.
type DelayedExpansion = 'on' | 'off'
const DELAYED_EXPANSION_STATES = ['off', 'on'] as const satisfies readonly DelayedExpansion[]
function runWrapper(
wrapperPath: string,
env: NodeJS.ProcessEnv,
// Why: `null` abandons stdin instead of closing it — the shape a caller outside an Orca
// pane produces, and the only way to prove the env guard exits before reading (#11549).
stdinPayload: string | null = PAYLOAD
stdinPayload: string | null = PAYLOAD,
delayedExpansion: DelayedExpansion = 'off'
): Promise<HookRun> {
return new Promise((resolve, reject) => {
// Why: mirror how Antigravity spawns the hook — `cmd /c <bare .cmd path>`, the exact
// chain in the bug report's process trace.
const child = spawn('cmd.exe', ['/d', '/c', wrapperPath], {
const child = spawn('cmd.exe', [`/v:${delayedExpansion}`, '/d', '/c', wrapperPath], {
stdio: ['pipe', 'pipe', 'pipe'],
windowsHide: true,
env
@@ -111,6 +118,7 @@ function runWrapper(
let timedOut = false
const timer = setTimeout(() => {
timedOut = true
child.stdin.destroy()
child.kill('SIGKILL')
}, 15_000)
child.on('error', (error) => {
@@ -154,6 +162,24 @@ function expectedStdout(eventName: string): string {
// Why: runs on every platform — the live delivery suite below is Windows-only, so this
// keeps a POSIX-only CI leg from letting the interpreter back into the hot path.
describe('Antigravity Windows hook post command', () => {
it.each(ANTIGRAVITY_EVENTS)('guards missing-core stdin for $eventName', ({ eventName }) => {
const script = getWindowsWrapperScript(eventName)
const drain = script.indexOf(WINDOWS_HOOK_STDIN_DRAIN_COMMAND)
const answer = script.lastIndexOf('echo {}')
expect(drain).toBeGreaterThan(answer)
for (const key of ['ORCA_AGENT_HOOK_PORT', 'ORCA_AGENT_HOOK_TOKEN', 'ORCA_PANE_KEY']) {
const guard = script.indexOf(`if "%${key}%"=="" exit /b 0`)
expect(guard, key).toBeGreaterThan(answer)
expect(guard, key).toBeLessThan(drain)
}
})
// Why (#9358/#9941): `%~dp0` carries the hooks path, so an inherited delayed expansion eats
// a `!` out of it and the wrapper silently misses the core on every event.
it.each(ANTIGRAVITY_EVENTS)('disables delayed expansion for $eventName', ({ eventName }) => {
expect(getWindowsWrapperScript(eventName)).toContain('setlocal DisableDelayedExpansion')
})
it('posts through curl.exe rather than a PowerShell interpreter', () => {
vi.spyOn(process, 'platform', 'get').mockReturnValue('win32')
const script = getManagedScript('local')
@@ -185,7 +211,10 @@ describe.skipIf(process.platform !== 'win32')('Antigravity Windows hook payload
})
it('delivers every event wrapper payload to the listener without spawning PowerShell', async () => {
home = mkdtempSync(join(tmpdir(), 'orca-antigravity-hook-'))
// Why the `!` in the directory: it lands in the wrapper's `%~dp0`, which is what an
// inherited delayed expansion eats (#9358/#9941). Without it the `/v:on` leg below
// proves nothing about the core lookup.
home = mkdtempSync(join(tmpdir(), 'orca-antigravity-hook!bang-'))
homedirMock.mockReturnValue(home)
expect(new AntigravityHookService().install().state).toBe('installed')
@@ -204,34 +233,42 @@ describe.skipIf(process.platform !== 'win32')('Antigravity Windows hook payload
ORCA_WORKTREE_ID: WORKTREE_ID
})
for (const event of ANTIGRAVITY_EVENTS) {
const label = event.eventName
const before = listener.posts.length
const result = await runWrapper(join(hooksDir, event.windowsWrapperFileName), env)
for (const delayedExpansion of DELAYED_EXPANSION_STATES) {
for (const event of ANTIGRAVITY_EVENTS) {
const label = `${event.eventName} (/v:${delayedExpansion})`
const before = listener.posts.length
const result = await runWrapper(
join(hooksDir, event.windowsWrapperFileName),
env,
PAYLOAD,
delayedExpansion
)
expect(result.timedOut, `${label} timed out`).toBe(false)
expect(result.exitCode, `${label} exit code`).toBe(0)
expect(result.stderr, `${label} stderr`).toBe('')
// Why: Antigravity reads silence on PreToolUse as deny (#2426), so the gate answer
// must survive the transport change.
expect(result.stdout.trim(), `${label} stdout`).toBe(expectedStdout(label))
expect(result.timedOut, `${label} timed out`).toBe(false)
expect(result.exitCode, `${label} exit code`).toBe(0)
expect(result.stderr, `${label} stderr`).toBe('')
// Why: Antigravity reads silence on PreToolUse as deny (#2426), so the gate answer
// must survive the transport change.
expect(result.stdout.trim(), `${label} stdout`).toBe(expectedStdout(event.eventName))
const posts = listener.posts.slice(before)
expect(posts, `${label} posted exactly one hook`).toHaveLength(1)
// Why: byte-exact, not "non-empty" — PowerShell recoded this body through the console
// code page, and a silently corrupted payload still looks posted.
expect(posts[0].payload, `${label} payload`).toBe(PAYLOAD)
expect(posts[0].hookEventName, `${label} hook_event_name`).toBe(label)
// Why: the `!` in both values is the delayed-expansion regression guard.
expect(posts[0].paneKey, `${label} paneKey`).toBe(PANE_KEY)
expect(posts[0].worktreeId, `${label} worktreeId`).toBe(WORKTREE_ID)
expect(posts[0].token, `${label} token`).toBe(HOOK_TOKEN)
expect(posts[0].contentType, `${label} content-type`).toContain(
'application/x-www-form-urlencoded'
)
const posts = listener.posts.slice(before)
expect(posts, `${label} posted exactly one hook`).toHaveLength(1)
// Why: byte-exact, not "non-empty" — PowerShell recoded this body through the console
// code page, and a silently corrupted payload still looks posted.
expect(posts[0].payload, `${label} payload`).toBe(PAYLOAD)
expect(posts[0].hookEventName, `${label} hook_event_name`).toBe(event.eventName)
// Why: the `!` in both values is the delayed-expansion regression guard — it is the
// `/v:on` leg that can actually fail on it.
expect(posts[0].paneKey, `${label} paneKey`).toBe(PANE_KEY)
expect(posts[0].worktreeId, `${label} worktreeId`).toBe(WORKTREE_ID)
expect(posts[0].token, `${label} token`).toBe(HOOK_TOKEN)
expect(posts[0].contentType, `${label} content-type`).toContain(
'application/x-www-form-urlencoded'
)
}
}
// Why: five wrapper launches plus a real install can overrun the default under load.
}, 60_000)
// Why: ten wrapper launches plus a real install can overrun the default under load.
}, 90_000)
// Why (#15117): Antigravity fires some events with no stdin at all. PowerShell substituted
// `{}` before posting; curl forwards the empty body, so prove the post still happens — the
@@ -264,6 +301,67 @@ describe.skipIf(process.platform !== 'win32')('Antigravity Windows hook payload
expect(listener.posts[0].hookEventName).toBe('PreInvocation')
}, 30_000)
// Why a helper: the missing-core cases all need a real install with the core removed, which
// is the shape an AV quarantine or a half-finished uninstall leaves behind.
async function installWithoutCore(): Promise<string> {
home = mkdtempSync(join(tmpdir(), 'orca-antigravity-fallback-'))
homedirMock.mockReturnValue(home)
expect(new AntigravityHookService().install().state).toBe('installed')
const hooksDir = join(home, '.orca', 'agent-hooks')
rmSync(join(hooksDir, 'antigravity-hook.cmd'))
return hooksDir
}
it.each(['ORCA_AGENT_HOOK_PORT', 'ORCA_AGENT_HOOK_TOKEN', 'ORCA_PANE_KEY'])(
'answers every missing-core event with abandoned stdin and no %s',
async (missingKey) => {
const hooksDir = await installWithoutCore()
const listener = await startHookListener()
server = listener.server
const env = hookEnvironment({
USERPROFILE: home,
HOME: home,
ORCA_AGENT_HOOK_PORT: String(listener.port),
ORCA_AGENT_HOOK_TOKEN: HOOK_TOKEN,
ORCA_PANE_KEY: PANE_KEY,
[missingKey]: ''
})
for (const event of ANTIGRAVITY_EVENTS) {
const result = await runWrapper(join(hooksDir, event.windowsWrapperFileName), env, null)
expect(result.timedOut, event.eventName).toBe(false)
expect(result.exitCode, event.eventName).toBe(0)
expect(result.stdout.trim(), event.eventName).toBe(expectedStdout(event.eventName))
expect(result.stderr, event.eventName).toBe('')
}
expect(listener.posts).toHaveLength(0)
},
90_000
)
// Why: the guard must not cost the valid path its drain — with the Orca env present the
// fallback still owns stdin, so the agent's payload write completes instead of breaking.
it('still drains a closed payload for every missing-core event inside a pane', async () => {
const hooksDir = await installWithoutCore()
const listener = await startHookListener()
server = listener.server
const env = hookEnvironment({
USERPROFILE: home,
HOME: home,
ORCA_AGENT_HOOK_PORT: String(listener.port),
ORCA_AGENT_HOOK_TOKEN: HOOK_TOKEN,
ORCA_PANE_KEY: PANE_KEY
})
for (const event of ANTIGRAVITY_EVENTS) {
const result = await runWrapper(join(hooksDir, event.windowsWrapperFileName), env)
expect(result.timedOut, event.eventName).toBe(false)
expect(result.exitCode, event.eventName).toBe(0)
expect(result.stdout.trim(), event.eventName).toBe(expectedStdout(event.eventName))
expect(result.stderr, event.eventName).toBe('')
}
// Why: the fallback answers the agent but has no core to post through.
expect(listener.posts).toHaveLength(0)
}, 60_000)
it('exits without reading stdin when the pane env is missing', async () => {
home = mkdtempSync(join(tmpdir(), 'orca-antigravity-hook-'))
homedirMock.mockReturnValue(home)
+6 -6
View File
@@ -71,7 +71,7 @@ function mapExternalRuns({
.map((run, index) => {
const runAt = asString(run.run_at) ?? asString(run.runAt)
const id = asString(run.id) ?? `${jobId}:${runAt ?? index}`
return {
const mapped: ExternalAutomationRun = {
id,
managerId,
provider,
@@ -83,15 +83,15 @@ function mapExternalRuns({
error: asString(run.error),
outputPath: asString(run.output_path) ?? asString(run.outputPath)
}
return { run: mapped, time: runAt ? Date.parse(runAt) : Number.NaN }
})
.sort((a, b) => {
const aTime = a.runAt ? Date.parse(a.runAt) : Number.NaN
const bTime = b.runAt ? Date.parse(b.runAt) : Number.NaN
if (Number.isFinite(aTime) && Number.isFinite(bTime)) {
return bTime - aTime
if (Number.isFinite(a.time) && Number.isFinite(b.time)) {
return b.time - a.time
}
return b.id.localeCompare(a.id)
return b.run.id.localeCompare(a.run.id)
})
.map(({ run }) => run)
}
function hermesScheduleDisplay(job: ExternalJobRecord): string {
@@ -0,0 +1,38 @@
import { expect, it, vi } from 'vitest'
import { mapHermesJobs, mapOpenClawJobs } from './external-job-mappers'
it.each([mapHermesJobs, mapOpenClawJobs])(
'parses run dates once and preserves provider fallback ordering',
(mapJobs) => {
const runs = Array.from({ length: 2000 }, (_, i) => ({
id: String(i),
run_at:
i % 137 === 0
? 'invalid'
: new Date(1700000000000 + ((i * 173) % 1999) * 1000).toISOString(),
output_content: `Output ${i}`,
status: 'completed'
}))
const parse = vi.spyOn(Date, 'parse')
let expected: typeof runs
let jobs: ReturnType<typeof mapHermesJobs>
try {
expected = [...runs].sort((a, b) => {
const left = Date.parse(a.run_at),
right = Date.parse(b.run_at)
return Number.isFinite(left) && Number.isFinite(right)
? right - left
: b.id.localeCompare(a.id)
})
expect(parse.mock.calls.length).toBeGreaterThan(10_000)
parse.mockClear()
jobs = mapJobs('manager', [{ id: 'job', runs }])
expect(parse).toHaveBeenCalledTimes(2000)
} finally {
parse.mockRestore()
}
expect(jobs[0].runs.map((run) => run.id)).toEqual(expected.map((run) => run.id))
expect(jobs[0].runs.every((run) => run.outputContent === `Output ${run.id}`)).toBe(true)
expect(jobs[0].runs.every((run) => !('time' in run))).toBe(true)
}
)
@@ -0,0 +1,117 @@
import { EventEmitter } from 'node:events'
import { createServer, type Socket } from 'node:net'
import { PassThrough } from 'node:stream'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { RemoteBrowserSocksServer } from './remote-browser-socks-server'
vi.mock('node:net', () => ({
createServer: vi.fn(() => ({ listening: false }))
}))
function setup(requestTail: Buffer = Buffer.alloc(0)) {
const upstream = new PassThrough()
const write = vi.spyOn(upstream, 'write')
const opened = Promise.withResolvers<PassThrough>()
const open = vi.fn(() => opened.promise)
const server = new RemoteBrowserSocksServer({ open })
const socket = Object.assign(new EventEmitter(), {
remoteAddress: '127.0.0.1',
destroyed: false,
write: vi.fn(() => true),
pause: vi.fn(),
end: vi.fn((_reply, callback) => callback()),
pipe: vi.fn(),
destroy: vi.fn(() => {
socket.destroyed = true
socket.emit('close')
})
})
const accept = vi.mocked(createServer).mock.calls.at(-1)![0] as (socket: Socket) => void
accept(socket as unknown as Socket)
socket.emit('data', Buffer.from([5, 1, 0]))
socket.emit('data', Buffer.concat([Buffer.from([5, 1, 0, 1, 127, 0, 0, 1, 1, 187]), requestTail]))
return { server, socket, upstream, write, opened, open }
}
afterEach(() => vi.restoreAllMocks())
describe('pending browser SOCKS route buffering', () => {
it('copies fragmented pending bytes linearly and forwards every byte at the existing cap', async () => {
const { server, socket, upstream, write, opened } = setup()
const payload = Buffer.alloc(256 * 1024)
for (let index = 0; index < payload.length; index += 1) {
payload[index] = index % 251
}
let copiedBytes = 0
const originalCopy = Buffer.prototype.copy
const copy = vi.spyOn(Buffer.prototype, 'copy').mockImplementation(function (target, ...args) {
const copied = originalCopy.call(this, target, ...args)
copiedBytes += copied
return copied
})
const concat = vi.spyOn(Buffer, 'concat')
try {
for (let index = 0; index < payload.length; index += 256) {
socket.emit('data', payload.subarray(index, index + 256))
}
expect(concat.mock.calls.length).toBe(0)
expect(copiedBytes).toBeLessThan(payload.length * 3)
opened.resolve(upstream)
await vi.waitFor(() => expect(write).toHaveBeenCalledTimes(1))
expect(write.mock.calls[0][0]).toEqual(payload)
} finally {
copy.mockRestore()
concat.mockRestore()
await server.close()
upstream.destroy()
}
})
it('keeps request-tail bytes ahead of later fragments in the pending payload', async () => {
const tail = Buffer.from('GET / HTTP/1.1\r\n')
const { server, socket, upstream, write, opened } = setup(tail)
const rest = Buffer.from('Host: example.com\r\n\r\n')
try {
for (const byte of rest) {
socket.emit('data', Buffer.from([byte]))
}
opened.resolve(upstream)
await vi.waitFor(() => expect(write).toHaveBeenCalledTimes(1))
expect(write.mock.calls[0][0]).toEqual(Buffer.concat([tail, rest]))
} finally {
await server.close()
upstream.destroy()
}
})
it('rejects one byte beyond the cap and destroys a late upstream without forwarding', async () => {
const { server, socket, upstream, write, opened, open } = setup()
try {
await vi.waitFor(() => expect(open).toHaveBeenCalledTimes(1))
socket.emit('data', Buffer.alloc(256 * 1024))
expect(socket.destroyed).toBe(false)
socket.emit('data', Buffer.from([1]))
expect(socket.destroyed).toBe(true)
expect(socket.end.mock.calls[0][0][1]).toBe(1)
opened.resolve(upstream)
await vi.waitFor(() => expect(upstream.destroyed).toBe(true))
expect(write).not.toHaveBeenCalled()
} finally {
await server.close()
}
})
it('discards pending input on client close while the route is opening', async () => {
const { server, socket, upstream, write, opened, open } = setup()
try {
await vi.waitFor(() => expect(open).toHaveBeenCalledTimes(1))
socket.emit('data', Buffer.from('pending request'))
socket.destroy()
opened.resolve(upstream)
await vi.waitFor(() => expect(upstream.destroyed).toBe(true))
expect(write).not.toHaveBeenCalled()
} finally {
await server.close()
}
})
})
+14 -21
View File
@@ -1,5 +1,7 @@
import { createServer, type Server, type Socket } from 'node:net'
import type { Duplex } from 'node:stream'
import { GrowingByteBuffer } from '../../shared/growing-byte-buffer'
import { pipeUpstreamToClient } from './remote-browser-socks-upstream'
const SOCKS_VERSION = 5
const SOCKS_NO_AUTH = 0
@@ -106,10 +108,13 @@ export class RemoteBrowserSocksServer {
this.clients.add(socket)
let phase: 'greeting' | 'request' | 'opening' | 'connected' | 'closed' = 'greeting'
let buffered = Buffer.alloc(0)
const pendingUpstream = new GrowingByteBuffer()
const timeout = setTimeout(() => socket.destroy(), HANDSHAKE_TIMEOUT_MS)
const cleanup = (): void => {
phase = 'closed'
clearTimeout(timeout)
buffered = Buffer.alloc(0)
pendingUpstream.clear()
this.clients.delete(socket)
}
const finishFailure = (reply: Uint8Array): void => {
@@ -119,6 +124,7 @@ export class RemoteBrowserSocksServer {
phase = 'closed'
clearTimeout(timeout)
buffered = Buffer.alloc(0)
pendingUpstream.clear()
socket.pause()
socket.end(reply, () => socket.destroy())
}
@@ -127,13 +133,15 @@ export class RemoteBrowserSocksServer {
if (phase === 'closed' || phase === 'connected') {
return
}
buffered = Buffer.concat([buffered, chunk])
if (phase === 'opening') {
if (buffered.byteLength > MAX_PENDING_UPSTREAM_BYTES) {
if (pendingUpstream.byteLength + chunk.byteLength > MAX_PENDING_UPSTREAM_BYTES) {
fail(1)
} else {
pendingUpstream.append(chunk)
}
return
}
buffered = Buffer.concat([buffered, chunk])
if (phase === 'greeting' && buffered.byteLength > MAX_HANDSHAKE_BYTES) {
fail(1)
return
@@ -181,6 +189,8 @@ export class RemoteBrowserSocksServer {
return
}
phase = 'opening'
pendingUpstream.append(buffered)
buffered = Buffer.alloc(0)
void Promise.resolve()
.then(() => this.open(normalizeListenerWildcard(parsed.target)))
.then(
@@ -193,9 +203,8 @@ export class RemoteBrowserSocksServer {
clearTimeout(timeout)
socket.off('data', onData)
socket.write(SUCCESS_RESPONSE)
if (buffered.byteLength > 0) {
upstream.write(buffered)
buffered = Buffer.alloc(0)
if (pendingUpstream.byteLength > 0) {
upstream.write(pendingUpstream.takeBuffer())
}
socket.pipe(upstream)
pipeUpstreamToClient(upstream, socket)
@@ -212,22 +221,6 @@ export class RemoteBrowserSocksServer {
}
}
function pipeUpstreamToClient(upstream: Duplex, socket: Socket): void {
upstream.on('data', (chunk: Buffer) => {
const bytes = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk)
const accepted = socket.write(bytes, (error) => {
if (!error && 'settleRead' in upstream && typeof upstream.settleRead === 'function') {
upstream.settleRead(bytes.byteLength)
}
})
if (!accepted) {
upstream.pause()
}
})
socket.on('drain', () => upstream.resume())
upstream.once('end', () => socket.end())
}
function parseSocksRequest(buffer: Uint8Array): SocksRequest | null | undefined {
if (buffer.byteLength < 4) {
return undefined
@@ -0,0 +1,18 @@
import type { Socket } from 'node:net'
import type { Duplex } from 'node:stream'
export function pipeUpstreamToClient(upstream: Duplex, socket: Socket): void {
upstream.on('data', (chunk: Buffer) => {
const bytes = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk)
const accepted = socket.write(bytes, (error) => {
if (!error && 'settleRead' in upstream && typeof upstream.settleRead === 'function') {
upstream.settleRead(bytes.byteLength)
}
})
if (!accepted) {
upstream.pause()
}
})
socket.on('drain', () => upstream.resume())
upstream.once('end', () => socket.end())
}
@@ -1,3 +1,4 @@
import { highestUsageKey } from '../usage/highest-usage-key'
import type {
ClaudeUsageBreakdownKind,
ClaudeUsageBreakdownRow,
@@ -56,9 +57,8 @@ export function buildSummary(
}
}
const topModel = [...byModel.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null
const topProject =
[...byProject.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null
const topModel = highestUsageKey(byModel)
const topProject = highestUsageKey(byProject)
return {
scope,
@@ -0,0 +1,53 @@
import { expect, it, vi } from 'vitest'
import { attributeClaudeUsageTurns } from './worktree-attribution'
import type { ClaudeUsageParsedTurn } from './types'
vi.mock('node:fs/promises', () => ({ realpath: async (path: string) => path }))
it('resolves repeated nested and unmatched cwd paths once per attribution batch', async () => {
const lookup = new Map(
Array.from({ length: 100 }, (_, index) => [
`/repo-${String(index).padStart(3, '0')}`,
{
repoId: `repo-${index}`,
worktreeId: `wt-${index}`,
path: `/repo-${index}`,
displayName: `Repo ${index}`
}
])
)
const input: ClaudeUsageParsedTurn[] = Array.from({ length: 1000 }, (_, index) => ({
sessionId: String(index),
timestamp: '2026-09-07T00:00:00Z',
model: null,
cwd: index % 2 === 0 ? '/repo-099/nested' : '/outside',
gitBranch: null,
inputTokens: 1,
outputTokens: 1,
cacheReadTokens: 0,
cacheWriteTokens: 0,
cacheWrite1hTokens: 0
}))
const original = String.prototype.startsWith
let comparisons = 0
const spy = vi.spyOn(String.prototype, 'startsWith').mockImplementation(function (
this: string,
search: string,
position?: number
) {
if (search.slice(0, 6) === '/repo-') {
comparisons += 1
}
return original.call(this, search, position)
})
let result: Awaited<ReturnType<typeof attributeClaudeUsageTurns>>
try {
result = await attributeClaudeUsageTurns(input, lookup)
} finally {
spy.mockRestore()
}
expect(comparisons).toBeLessThanOrEqual(200)
expect(result![0].worktreeId).toBe('wt-99')
expect(result![1].worktreeId).toBeNull()
expect(result![1].projectKey).toBe('cwd:/outside')
})
@@ -105,7 +105,7 @@ export async function attributeClaudeUsageTurns(
worktreeLookup: Map<string, ClaudeUsageWorktreeRef>
): Promise<ClaudeUsageAttributedTurn[]> {
const attributed: ClaudeUsageAttributedTurn[] = []
const canonicalCwdByPath = new Map<string, string>()
const worktreeByCwd = new Map<string, ClaudeUsageWorktreeRef | null>()
for (const turn of turns) {
const day = localDayFromTimestamp(turn.timestamp)
@@ -119,14 +119,12 @@ export async function attributeClaudeUsageTurns(
let projectLabel = getDefaultProjectLabel(turn.cwd)
if (turn.cwd) {
let canonicalCwd = canonicalCwdByPath.get(turn.cwd)
if (canonicalCwd === undefined) {
// Why: Claude transcripts repeat the same cwd for many consecutive
// turns. Cache realpath work so attribution scales with unique paths.
canonicalCwd = await canonicalizePath(turn.cwd)
canonicalCwdByPath.set(turn.cwd, canonicalCwd)
let worktree = worktreeByCwd.get(turn.cwd)
if (worktree === undefined) {
const canonicalCwd = await canonicalizePath(turn.cwd)
worktree = findContainingWorktree(canonicalCwd, worktreeLookup)
worktreeByCwd.set(turn.cwd, worktree)
}
const worktree = findContainingWorktree(canonicalCwd, worktreeLookup)
if (worktree) {
repoId = worktree.repoId
worktreeId = worktree.worktreeId
@@ -20,14 +20,22 @@ function record(value: unknown): Record<string, unknown> | null {
return typeof value === 'object' && value !== null ? (value as Record<string, unknown>) : null
}
function taskId(message: Record<string, unknown>): string | null {
const value = message.task_id
return typeof value === 'string' && value.length > 0 && value.length <= MAX_TASK_ID_LENGTH
? value
: null
/** The bound every task id shares, wherever it enters. An id the roster stores
* becomes a durable entry key, so a provisional one takes the same bound the
* announced path applies — an over-long id is rejected, never truncated. */
export function isBoundedClaudeTaskId(value: string): boolean {
return value.length > 0 && value.length <= MAX_TASK_ID_LENGTH
}
function taskDescription(value: unknown): string | undefined {
/** The task's canonical, resume-stable id. Shared with the subagent roster so
* both readers of this channel agree on what identifies a task. */
export function claudeTaskId(message: Record<string, unknown>): string | null {
const value = message.task_id
return typeof value === 'string' && isBoundedClaudeTaskId(value) ? value : null
}
/** A task's human label, collapsed and bounded. */
export function claudeTaskDescription(value: unknown): string | undefined {
if (typeof value !== 'string') {
return undefined
}
@@ -107,7 +115,7 @@ export class ClaudeBackgroundTaskTracker {
this.replaceAggregateRoster(message.tasks)
return true
}
const id = taskId(message)
const id = claudeTaskId(message)
if (!id) {
return false
}
@@ -126,13 +134,13 @@ export class ClaudeBackgroundTaskTracker {
}
const existing = this.tasks.get(id)
if (
(patch.is_backgrounded === true || taskDescription(patch.description)) &&
(patch.is_backgrounded === true || claudeTaskDescription(patch.description)) &&
(!this.aggregateRosterObserved || existing)
) {
this.upsert(id, {
backgrounded: patch.is_backgrounded === true || existing?.backgrounded === true,
kind: existing?.kind ?? 'unknown',
description: taskDescription(patch.description) ?? existing?.description
description: claudeTaskDescription(patch.description) ?? existing?.description
})
return true
}
@@ -152,7 +160,7 @@ export class ClaudeBackgroundTaskTracker {
this.upsert(id, {
backgrounded: message.is_backgrounded === true || kind === 'workflow' || kind === 'monitor',
kind,
description: taskDescription(message.description)
description: claudeTaskDescription(message.description)
})
return true
}
@@ -172,14 +180,14 @@ export class ClaudeBackgroundTaskTracker {
if (!task || task.ambient === true) {
continue
}
const id = taskId(task)
const id = claudeTaskId(task)
if (!id) {
continue
}
this.tasks.set(id, {
backgrounded: true,
kind: classifyClaudeBackgroundTaskKind(task.task_type),
description: taskDescription(task.description)
description: claudeTaskDescription(task.description)
})
}
}
@@ -0,0 +1,165 @@
import { mkdtemp, rm, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types'
import {
claudeDispatchInvokesSlashCommand,
claudeDispatchMessageContent
} from './claude-structured-dispatch-content'
const PNG = Buffer.from(
'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==',
'base64'
)
function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem {
return { kind: 'message', role: 'user', blocks }
}
const REMOTE_IMAGE = { type: 'image-ref' as const, url: 'https://example.test/a.png' }
describe('claudeDispatchMessageContent', () => {
it('puts the text block last so a slash command still expands with an attachment', async () => {
const content = await claudeDispatchMessageContent(
// The composer builds text-then-images; Claude only treats a leading `/` as a
// command when the LAST block is text.
userMessage([{ type: 'text', text: '/goal ship the parser' }, REMOTE_IMAGE])
)
expect(content).toEqual([
{ type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } },
{ type: 'text', text: '/goal ship the parser' }
])
})
it('keeps every image ahead of the text and preserves each side’s order', async () => {
const second = { type: 'image-ref' as const, url: 'https://example.test/b.png' }
const content = await claudeDispatchMessageContent(
userMessage([{ type: 'text', text: 'look' }, REMOTE_IMAGE, second])
)
expect(content.map((part) => (part as { type: string }).type)).toEqual([
'image',
'image',
'text'
])
expect(content[0]).toEqual({
type: 'image',
source: { type: 'url', url: 'https://example.test/a.png' }
})
expect(content[1]).toEqual({
type: 'image',
source: { type: 'url', url: 'https://example.test/b.png' }
})
})
it('sends text alone unchanged', async () => {
const content = await claudeDispatchMessageContent(userMessage([{ type: 'text', text: 'hi' }]))
expect(content).toEqual([{ type: 'text', text: 'hi' }])
})
it('sends an image with no text', async () => {
const content = await claudeDispatchMessageContent(userMessage([REMOTE_IMAGE]))
expect(content).toEqual([
{ type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } }
])
})
it('rejects a message with no renderable block', async () => {
await expect(
claudeDispatchMessageContent(userMessage([{ type: 'text', text: '' }]))
).rejects.toThrow('Claude dispatch requires text or an image')
})
it('rejects a non-user message', async () => {
await expect(
claudeDispatchMessageContent({
...userMessage([{ type: 'text', text: 'hi' }]),
role: 'assistant'
})
).rejects.toThrow('Claude dispatch accepts only user messages')
})
it('joins several text blocks so a command is not stranded ahead of trailing prose', async () => {
// Appending each block would leave `thanks` trailing, and Claude reads only that block.
const content = await claudeDispatchMessageContent(
userMessage([
{ type: 'text', text: '/goal ship' },
REMOTE_IMAGE,
{ type: 'text', text: 'thanks' }
])
)
expect(content).toEqual([
{ type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } },
{ type: 'text', text: '/goal ship\nthanks' }
])
expect(claudeDispatchInvokesSlashCommand(content)).toBe(true)
})
it('puts a locally attached image ahead of the text, the shape the composer sends', async () => {
const dir = await mkdtemp(join(tmpdir(), 'claude-dispatch-content-'))
const path = join(dir, 'shot.png')
await writeFile(path, PNG)
try {
const content = await claudeDispatchMessageContent(
userMessage([
{ type: 'text', text: '/goal ship' },
{ type: 'image-ref', path }
])
)
expect(content).toEqual([
{
type: 'image',
source: { type: 'base64', media_type: 'image/png', data: PNG.toString('base64') }
},
{ type: 'text', text: '/goal ship' }
])
} finally {
await rm(dir, { recursive: true, force: true })
}
})
})
describe('claudeDispatchInvokesSlashCommand', () => {
it('reads the trailing prompt Claude recovers, not any text block', () => {
expect(
claudeDispatchInvokesSlashCommand([
{ type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } },
{ type: 'text', text: '/goal ship' }
])
).toBe(true)
// The pre-fix order: Claude recovers no prompt at all, so no command runs.
expect(
claudeDispatchInvokesSlashCommand([
{ type: 'text', text: '/goal ship' },
{ type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } }
])
).toBe(false)
})
it('reads the joined prompt, so a command behind leading prose is not one', async () => {
// Keeping the blocks separate would leave `/goal ship` trailing and falsely claim a command.
const content = await claudeDispatchMessageContent(
userMessage([
{ type: 'text', text: 'take a look' },
{ type: 'text', text: '/goal ship' }
])
)
expect(content).toEqual([{ type: 'text', text: 'take a look\n/goal ship' }])
expect(claudeDispatchInvokesSlashCommand(content)).toBe(false)
})
it('matches untrimmed, as Claude does, and ignores a promptless turn', () => {
expect(claudeDispatchInvokesSlashCommand([{ type: 'text', text: ' /goal ship' }])).toBe(false)
expect(claudeDispatchInvokesSlashCommand([{ type: 'text', text: 'ship it' }])).toBe(false)
expect(claudeDispatchInvokesSlashCommand([])).toBe(false)
})
})
@@ -3,6 +3,7 @@ import { open } from 'node:fs/promises'
import { extname } from 'node:path'
import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types'
import type { NativeChatBlock } from '../../shared/native-chat-types'
import { claudeRecord } from './claude-structured-item-translation'
const MAX_IMAGE_BYTES = 5 * 1024 * 1024
const MAX_IMAGE_COUNT = 20
@@ -88,27 +89,49 @@ async function imageContent(
}
}
/**
* Claude encodes a user turn as attachment blocks followed by the typed text, and recovers the
* typed prompt by reading only the trailing text block. Verified against the real CLI over
* stream-json: a body ending in an image has no recoverable prompt, so its `/command` reaches
* the model as prose instead of being expanded.
*/
export async function claudeDispatchMessageContent(
body: AgentJournalMessageItem
): Promise<unknown[]> {
if (body.role !== 'user') {
throw new Error('Claude dispatch accepts only user messages')
}
const content: unknown[] = []
const images: unknown[] = []
const texts: string[] = []
const imageBudget: ImageBudget = { count: 0, localBytes: 0 }
for (const block of body.blocks as NativeChatBlock[]) {
if (block.type === 'text' && block.text.length > 0) {
content.push({ type: 'text', text: block.text })
texts.push(block.text)
} else if (block.type === 'image-ref') {
content.push(await imageContent(block, imageBudget))
images.push(await imageContent(block, imageBudget))
}
}
// Join rather than append each block: only the trailing text is read as the prompt, so several
// text blocks would silently discard every one but the last.
const content = texts.length > 0 ? [...images, { type: 'text', text: texts.join('\n') }] : images
if (content.length === 0) {
throw new Error('Claude dispatch requires text or an image')
}
return content
}
/** The prompt Claude recovers from a dispatch, or null when the turn carries no prompt. */
function claudeDispatchPrompt(content: readonly unknown[]): string | null {
const last = claudeRecord(content.at(-1))
return last?.type === 'text' && typeof last.text === 'string' ? last.text : null
}
/** Mirrors how Claude decides a turn is a command. Untrimmed on purpose: Claude does not trim
* here either, so leading whitespace really does mean no command runs. */
export function claudeDispatchInvokesSlashCommand(content: readonly unknown[]): boolean {
return claudeDispatchPrompt(content)?.startsWith('/') === true
}
/**
* Keep waiter metadata bounded even when a dispatch contains large base64 images.
* The digest is only diagnostic: replay acknowledgement must use provider identity.
@@ -117,10 +140,7 @@ export function claudeDispatchContentKey(content: readonly unknown[]): string {
const digest = createHash('sha256')
const summary = content
.map((part) => {
const record =
typeof part === 'object' && part !== null && !Array.isArray(part)
? (part as Record<string, unknown>)
: null
const record = claudeRecord(part)
const type = typeof record?.type === 'string' ? record.type : 'unknown'
if (type === 'text') {
return `text:${typeof record?.text === 'string' ? record.text.length : 0}`
@@ -136,10 +156,7 @@ export function claudeDispatchContentKey(content: readonly unknown[]): string {
})
.join(',')
for (const [index, part] of content.entries()) {
const record =
typeof part === 'object' && part !== null && !Array.isArray(part)
? (part as Record<string, unknown>)
: null
const record = claudeRecord(part)
const type = typeof record?.type === 'string' ? record.type : 'unknown'
digest.update(`${index}:${type}:`)
if (type === 'text' && typeof record?.text === 'string') {
@@ -423,6 +423,73 @@ describe('Claude structured dispatch image limits', () => {
})
})
it('accepts a slash command sent with an attachment from its result receipt', async () => {
const session = sessionFor()
const dispatched = dispatchClaudeTurn(
session,
{
clientMessageId: 'client-1',
body: userMessage([
{ type: 'text', text: '/permissions' },
{ type: 'image-ref', url: 'https://example.test/a.png' }
])
},
100
)
await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1))
// The mapper moves the image ahead of the prompt, so Claude runs the command and replies
// with a result receipt instead of a user replay.
expect(
resolveClaudeReplayWaiter(session, {
type: 'result',
subtype: 'success',
session_id: 'provider-session',
uuid: 'command-result-uuid'
})
).toBe(false)
await expect(dispatched).resolves.toMatchObject({
state: 'accepted',
providerIdentity: { uuid: 'command-result-uuid' }
})
// The sent order is the fix: the waiter's verdict alone was already what it is today.
expect(session.connection.send).toHaveBeenCalledWith(
expect.objectContaining({
message: {
role: 'user',
content: [
{ type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } },
{ type: 'text', text: '/permissions' }
]
}
})
)
})
it('does not take a result receipt for leading whitespace Claude never reads as a command', async () => {
const session = sessionFor()
const dispatched = dispatchClaudeTurn(
session,
{
clientMessageId: 'client-1',
body: userMessage([{ type: 'text', text: ' /permissions' }])
},
100
)
await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1))
expect(
resolveClaudeReplayWaiter(session, {
type: 'result',
subtype: 'success',
session_id: 'provider-session',
uuid: 'unrelated-result-uuid'
})
).toBe(false)
await expect(dispatched).resolves.toMatchObject({ state: 'unknown' })
})
it('correlates a later slash-command result by user_message_uuid despite a timed-out slash waiter', async () => {
const session = sessionFor()
const first = dispatchClaudeTurn(
@@ -12,6 +12,7 @@ import type { ClaudeDispatchWaiter, ClaudeSession } from './claude-structured-se
import { readClaudeFrameString } from './claude-structured-init-proof'
import {
claudeDispatchContentKey,
claudeDispatchInvokesSlashCommand,
claudeDispatchMessageContent
} from './claude-structured-dispatch-content'
@@ -231,9 +232,9 @@ export async function dispatchClaudeTurn(
return { state: 'rejected', reason: (error as Error).message }
}
const dispatchSequence = ++session.dispatchSequence
const acceptsResult = input.body.blocks.some(
(block) => block.type === 'text' && block.text.trimStart().startsWith('/')
)
// Read the sent content, not the journal blocks: only the mapped trailing prompt decides
// whether Claude runs a command, so the two cannot disagree about which frame settles this.
const acceptsResult = claudeDispatchInvokesSlashCommand(content)
const sentUuid = randomUUID()
const replay = waitForReplay(
session,
@@ -74,6 +74,18 @@ export function claudeMessageIdentity(
return { provider: 'claude', sessionId: envelope.sessionId, uuid: envelope.uuid }
}
/** User bubbles belong to the submitted message; SDK user frames carry echoes
* and tool results, so a user envelope keeps only its tool results. */
export function claudeOutputEnvelope(envelope: ClaudeMessageEnvelope): ClaudeMessageEnvelope {
if (envelope.role !== 'user') {
return envelope
}
return {
...envelope,
content: envelope.content.filter((part) => claudeRecord(part)?.type === 'tool_result')
}
}
function messageBlocks(envelope: ClaudeMessageEnvelope): NativeChatBlock[] {
const blocks: NativeChatBlock[] = []
for (const value of envelope.content) {
@@ -0,0 +1,259 @@
import { describe, expect, it, vi } from 'vitest'
import type {
AgentJournalItemBody,
AgentJournalItemIdentity
} from '../../shared/agent-session-journal-types'
import type {
NativeChatSubagentEntry,
NativeChatSubagentGroupBlock
} from '../../shared/native-chat-types'
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
import { createClaudeJournalTranslator } from './claude-structured-journal-translation'
const GROUP_ITEM_ID = 'claude-subagents:claude-session:user-1'
/** The union's other arms carry no client message id, so reading one narrows. */
function orcaClientMessageId(identity: AgentJournalItemIdentity): string | null {
return identity.provider === 'orca' ? identity.clientMessageId : null
}
function harness() {
const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = []
const sink: StructuredAgentSessionEventSink = {
appendItem: (identity, body) => items.push({ identity, body }),
appendTombstone: vi.fn(),
publish: vi.fn()
}
const translator = createClaudeJournalTranslator({ sink, fallbackIdPrefix: 'test' })
const groupRows = () =>
items.filter((item) => orcaClientMessageId(item.identity) === GROUP_ITEM_ID)
const agentsOf = (body: AgentJournalItemBody | undefined): NativeChatSubagentEntry[] => {
if (!body || body.kind !== 'message') {
return []
}
const block = body.blocks.find(
(candidate): candidate is NativeChatSubagentGroupBlock => candidate.type === 'subagent-group'
)
return block ? block.agents : []
}
/** The last roster row written for one group, so a test can read a group that
* is no longer the live one. */
const rosterIn = (groupId: string): NativeChatSubagentEntry[] =>
agentsOf(
items.findLast((item) => orcaClientMessageId(item.identity) === `claude-subagents:${groupId}`)
?.body
)
const rosterOf = (turnUuid: string): NativeChatSubagentEntry[] =>
rosterIn(`claude-session:${turnUuid}`)
const roster = (): NativeChatSubagentEntry[] => agentsOf(groupRows().at(-1)?.body)
const fallbackRows = (): AgentJournalItemBody[] =>
items
.filter((item) => (orcaClientMessageId(item.identity) ?? '').startsWith('provider-frame:'))
.map((item) => item.body)
return { translator, groupRows, roster, rosterIn, rosterOf, fallbackRows }
}
function userTurn(uuid: string) {
return {
type: 'message' as const,
sessionId: 'orca-session',
startsTurn: true as const,
message: {
type: 'user',
uuid,
session_id: 'claude-session',
parent_tool_use_id: null,
message: { role: 'user', content: [{ type: 'text', text: 'go' }] }
}
}
}
function systemFrame(subtype: string, fields: Record<string, unknown>) {
return {
type: 'message' as const,
sessionId: 'orca-session',
message: { type: 'system', subtype, session_id: 'claude-session', ...fields }
}
}
function spawnResult(uuid: string, toolUseId: string) {
return {
type: 'message' as const,
sessionId: 'orca-session',
message: {
type: 'user',
uuid,
session_id: 'claude-session',
parent_tool_use_id: null,
message: {
role: 'user',
content: [{ type: 'tool_result', tool_use_id: toolUseId, content: 'done' }]
}
}
}
}
function resultFrame() {
return {
type: 'message' as const,
sessionId: 'orca-session',
message: {
type: 'result',
subtype: 'success',
session_id: 'claude-session',
uuid: 'result-1',
result: 'ok'
}
}
}
describe('claude journal translation — subagents', () => {
it('rosters a spawned subagent and settles it on the spawn call result', () => {
const { translator, roster, fallbackRows } = harness()
translator.handle(userTurn('user-1'))
translator.handle(
systemFrame('task_started', {
task_id: 'task-1',
tool_use_id: 'toolu_1',
task_type: 'local_agent',
subagent_type: 'explorer',
description: 'Map the lane'
})
)
expect(roster()).toEqual([
expect.objectContaining({ id: 'task-1', label: 'Map the lane', state: 'working' })
])
// The task frames stay status-chrome, so none of them prints an opcode row.
expect(fallbackRows()).toEqual([])
translator.handle(spawnResult('user-2', 'toolu_1'))
expect(roster()).toEqual([expect.objectContaining({ state: 'completed' })])
})
it('marks a child still working at turn end unverifiable', () => {
const { translator, roster } = harness()
translator.handle(userTurn('user-1'))
translator.handle(
systemFrame('task_started', {
task_id: 'task-1',
task_type: 'local_agent',
description: 'Map the lane'
})
)
translator.handle(resultFrame())
expect(roster()).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
it('leaves a backgrounded child running past the end of its turn', () => {
const { translator, roster } = harness()
translator.handle(userTurn('user-1'))
translator.handle(
systemFrame('task_started', {
task_id: 'task-1',
tool_use_id: 'toolu_1',
task_type: 'local_agent',
description: 'Watch the build',
is_backgrounded: true
})
)
// A backgrounded spawn returns its tool result immediately; the child runs on.
translator.handle(spawnResult('user-2', 'toolu_1'))
translator.handle(resultFrame())
expect(roster()).toEqual([expect.objectContaining({ state: 'working' })])
translator.handle({ type: 'ended', sessionId: 'orca-session', reason: 'closed' })
expect(roster()).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
it('keeps a backgrounded shell task out of the roster entirely', () => {
const { translator, groupRows } = harness()
translator.handle(userTurn('user-1'))
translator.handle(
systemFrame('task_started', {
task_id: 'task-bash',
tool_use_id: 'toolu_bash',
task_type: 'local_bash',
description: 'sleep 20',
is_backgrounded: true
})
)
translator.handle(resultFrame())
expect(groupRows()).toEqual([])
})
it('shows a subagent whose release announces no task frames, from its child traffic', () => {
const { translator, roster } = harness()
translator.handle(userTurn('user-1'))
translator.handle({
type: 'message' as const,
sessionId: 'orca-session',
message: {
type: 'assistant',
uuid: 'child-1',
session_id: 'claude-session',
parent_tool_use_id: 'toolu_1',
message: { role: 'assistant', content: [{ type: 'text', text: 'looking' }] }
}
})
expect(roster()).toEqual([
expect.objectContaining({ id: 'toolu_1', label: 'subagent', state: 'working' })
])
})
it('settles the turn a new turn superseded, and leaves the new one running', () => {
const { translator, rosterOf } = harness()
translator.handle(userTurn('user-1'))
translator.handle(
systemFrame('task_started', {
task_id: 'task-1',
task_type: 'local_agent',
description: 'First turn'
})
)
// A second turn starts with no result frame for the first: the first turn
// ends here, and nothing else will ever name its group again.
translator.handle(userTurn('user-2'))
translator.handle(
systemFrame('task_started', {
task_id: 'task-2',
task_type: 'local_agent',
description: 'Second turn'
})
)
expect(rosterOf('user-1')).toEqual([expect.objectContaining({ state: 'unverifiable' })])
expect(rosterOf('user-2')).toEqual([expect.objectContaining({ state: 'working' })])
})
it('does not let an unrelated turn end settle a child announced outside a turn', () => {
const { translator, rosterIn } = harness()
// No turn is live yet, so this child has no turn key to belong to.
translator.handle(
systemFrame('task_started', {
task_id: 'task-early',
task_type: 'local_agent',
description: 'Before the turn'
})
)
translator.handle(userTurn('user-1'))
translator.handle(resultFrame())
expect(rosterIn('outside-turn')).toEqual([expect.objectContaining({ state: 'working' })])
// The outcome still lands, which a latched `unverifiable` would have lost.
translator.handle(
systemFrame('task_updated', { task_id: 'task-early', patch: { status: 'completed' } })
)
expect(rosterIn('outside-turn')).toEqual([expect.objectContaining({ state: 'completed' })])
})
it('settles a child left outside every turn when the session ends', () => {
const { translator, rosterIn } = harness()
translator.handle(
systemFrame('task_started', {
task_id: 'task-early',
task_type: 'local_agent',
description: 'Before the turn'
})
)
translator.handle(userTurn('user-1'))
translator.handle(resultFrame())
translator.handle({ type: 'ended', sessionId: 'orca-session', reason: 'closed' })
expect(rosterIn('outside-turn')).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
})
@@ -11,9 +11,8 @@ import {
claudeMessageBody,
claudeMessageIdentity,
claudeHasReplayContent,
claudeRecord,
claudeOutputEnvelope,
claudeStreamingMessageBody,
claudeText,
claudeThinkingIdentity,
claudeThinkingText,
claudeToolBody,
@@ -29,16 +28,15 @@ import {
claudeQuestionItems
} from './claude-structured-prompt-items'
import type { ClaudePromptRegistry } from './claude-structured-prompt-replies'
import { readableProviderFrameText } from '../native-chat/agent-session-wire/unhandled-provider-frame'
import { claudeProviderFrameActivity } from '../native-chat/agent-session-wire/provider-frame-activity'
import {
CLAUDE_UNRENDERABLE_CONTENT_TEXT,
appendUnmodeledClaudeContent,
claudeProviderFrameKind,
claudeResultFailure,
createClaudeProviderFrameFallback,
isModeledClaudeContent,
isSettledClaudeResultKind
} from './claude-structured-provider-fallback'
import { ClaudeSubagentRoster } from './claude-subagent-roster'
import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity'
import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints'
@@ -89,10 +87,16 @@ export function createClaudeJournalTranslator(
const promptItems = new Map<string, AgentJournalItemIdentity[]>()
const streamedBlocks = createClaudeStreamedBlockRegistry()
let currentTurn: { sessionId: string; turnId: string } | null = null
const groupKeyOf = (turn: { sessionId: string; turnId: string } | null): string | null =>
turn ? `${turn.sessionId}:${turn.turnId}` : null
const providerFallback = createClaudeProviderFrameFallback(
deps.sink,
deps.fallbackIdPrefix ?? 'acquisition'
)
const subagents = new ClaudeSubagentRoster({
sink: deps.sink,
currentGroupKey: () => groupKeyOf(currentTurn)
})
const streamedText = createClaudeStreamedTextCheckpoints({
...(deps.coalesceMs === undefined ? {} : { coalesceMs: deps.coalesceMs }),
...(deps.schedule ? { schedule: deps.schedule } : {}),
@@ -144,14 +148,10 @@ export function createClaudeJournalTranslator(
return false
}
let changed = false
// User bubbles belong to the submitted message; SDK user frames carry echoes and tool results.
const outputEnvelope =
envelope.role === 'user'
? {
...envelope,
content: envelope.content.filter((part) => claudeRecord(part)?.type === 'tool_result')
}
: envelope
if (envelope.parentToolUseId) {
subagents.observeChildActivity(envelope.parentToolUseId)
}
const outputEnvelope = claudeOutputEnvelope(envelope)
const body = claudeMessageBody(outputEnvelope)
// The final frame of a streamed block lands on the block's identity, not its own uuid.
const identity =
@@ -180,6 +180,8 @@ export function createClaudeJournalTranslator(
claudeToolIdentity(envelope.sessionId, result.toolUseId),
claudeToolBody({ tool, result })
)
// A spawn call's result is the parent turn's evidence its child finished.
subagents.observeToolResult(result.toolUseId, result.failed)
// Tool inputs are only needed until their matching result arrives.
tools.delete(result.toolUseId)
changed = true
@@ -192,21 +194,7 @@ export function createClaudeJournalTranslator(
})
changed = true
}
const unhandledContent = outputEnvelope.content.filter((part) => !isModeledClaudeContent(part))
for (const part of unhandledContent) {
const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown'
providerFallback.append(
`message:${envelope.role}:content:${partType}`,
part,
readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT
)
changed = true
}
// An empty user frame is a replay with nothing to show, not an unknown kind.
if (envelope.content.length === 0 && envelope.role === 'assistant') {
providerFallback.append(`message:${envelope.role}:empty`, message)
changed = true
}
changed = appendUnmodeledClaudeContent(providerFallback, outputEnvelope, message) || changed
if (
envelope.role === 'user' &&
startsTurn &&
@@ -214,6 +202,9 @@ export function createClaudeJournalTranslator(
message.parent_tool_use_id === null
) {
if (currentTurn) {
// A new turn starting is the only end the previous one gets when its
// result never arrives; settling it later would sweep THIS turn.
subagents.settleTurn(groupKeyOf(currentTurn))
publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false)
}
currentTurn = { sessionId: envelope.sessionId, turnId: envelope.uuid }
@@ -254,6 +245,8 @@ export function createClaudeJournalTranslator(
handle: (event) => {
if (event.type === 'ended') {
streamedText.flush()
// No event will ever settle a child once the provider is gone.
subagents.settleSession()
if (currentTurn) {
publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false)
currentTurn = null
@@ -274,6 +267,9 @@ export function createClaudeJournalTranslator(
promptItems.delete(event.promptKey)
deps.sink.publish()
} else if (event.type === 'message' && event.message.type === 'result') {
// The turn is over however it ended, so a foreground child still
// reported as working will never be settled by an event.
subagents.settleTurn(groupKeyOf(currentTurn))
if (currentTurn) {
publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false)
currentTurn = null
@@ -291,6 +287,9 @@ export function createClaudeJournalTranslator(
providerFallback.append(kind, event.message, failure?.text)
}
} else if (event.type === 'message') {
// These frames stay `status-chrome`: the roster reads them here, and the
// fallback below still drops the raw frame instead of printing an opcode.
subagents.observeSystemFrame(event.message)
const kind = claudeProviderFrameKind(event.message)
if (!handleMessage(event.message, event.startsTurn === true)) {
providerFallback.append(kind, event.message)
@@ -310,6 +309,7 @@ export function createClaudeJournalTranslator(
tools.clear()
promptItems.clear()
streamedBlocks.clear()
subagents.dispose()
}
}
}
@@ -4,8 +4,15 @@ import {
DEFAULT_JOURNAL_PAYLOAD_LIMITS
} from '../native-chat/agent-session-journal/journal-payload-bounds'
import { CLAUDE_STREAM_JSON_FRAME_KINDS } from '../native-chat/agent-session-wire/claude-stream-json-frame-schema'
import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame'
import { claudeRecord, claudeText } from './claude-structured-item-translation'
import {
readableProviderFrameText,
unhandledProviderFrameJournalItem
} from '../native-chat/agent-session-wire/unhandled-provider-frame'
import {
claudeRecord,
claudeText,
type ClaudeMessageEnvelope
} from './claude-structured-item-translation'
export function claudeProviderFrameKind(message: Record<string, unknown>): string {
const type = claudeText(message.type) ?? 'unknown'
@@ -123,3 +130,30 @@ export function createClaudeProviderFrameFallback(
}
}
}
export type ClaudeProviderFrameFallback = ReturnType<typeof createClaudeProviderFrameFallback>
/** Journal each content part this build does not model, plus the empty assistant
* frame a replay leaves behind (an empty USER frame is a replay with nothing to
* show, not an unknown kind). Returns whether anything was appended. */
export function appendUnmodeledClaudeContent(
fallback: ClaudeProviderFrameFallback,
envelope: ClaudeMessageEnvelope,
message: Record<string, unknown>
): boolean {
let changed = false
for (const part of envelope.content.filter((part) => !isModeledClaudeContent(part))) {
const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown'
fallback.append(
`message:${envelope.role}:content:${partType}`,
part,
readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT
)
changed = true
}
if (envelope.content.length === 0 && envelope.role === 'assistant') {
fallback.append(`message:${envelope.role}:empty`, message)
changed = true
}
return changed
}
@@ -0,0 +1,57 @@
import { describe, expect, it } from 'vitest'
import type { NativeChatSubagentEntry } from '../../shared/native-chat-types'
import { claudeSubagentGroupBody } from './claude-subagent-group-row'
function entry(id: string, state: NativeChatSubagentEntry['state']): NativeChatSubagentEntry {
return { id, label: id, state, startedAt: 1 }
}
/** The fallback sentence is the WHOLE row on mobile and paired web, which have
* no roster renderer, so these assertions are the entire contract there. */
function sentence(agents: readonly NativeChatSubagentEntry[]): string {
const body = claudeSubagentGroupBody('turn-1', agents)
const block = body.kind === 'message' ? body.blocks[0] : undefined
return block && block.type === 'text' ? block.text : ''
}
describe('claudeSubagentGroupBody fallback sentence', () => {
it('reads as a plain completion when every child completed', () => {
expect(sentence([entry('a', 'completed'), entry('b', 'completed')])).toBe('Ran 2 subagents')
})
it('keeps the singular noun for a lone child', () => {
expect(sentence([entry('a', 'completed')])).toBe('Ran 1 subagent')
expect(sentence([entry('a', 'working')])).toBe('Kicked off 1 subagent')
})
it('names an unverifiable child instead of claiming the group ran', () => {
expect(sentence([entry('a', 'completed'), entry('b', 'unverifiable')])).toBe(
'Ran 2 subagents (1 unverifiable)'
)
})
it('ranks the adverse outcome worst-first', () => {
expect(
sentence([entry('a', 'failed'), entry('b', 'unverifiable'), entry('c', 'completed')])
).toBe('Ran 3 subagents (1 failed)')
expect(sentence([entry('a', 'stopped'), entry('b', 'unverifiable')])).toBe(
'Ran 2 subagents (1 stopped)'
)
})
it('shows the adverse outcome while a sibling still works', () => {
expect(
sentence([entry('a', 'working'), entry('b', 'working'), entry('c', 'unverifiable')])
).toBe('Kicked off 3 subagents (1 unverifiable)')
})
it('leaves a benign settled state out of the sentence', () => {
expect(sentence([entry('a', 'idle'), entry('b', 'completed')])).toBe('Ran 2 subagents')
})
it('counts every child holding the worst adverse state', () => {
expect(sentence([entry('a', 'failed'), entry('b', 'failed'), entry('c', 'stopped')])).toBe(
'Ran 3 subagents (2 failed)'
)
})
})
@@ -0,0 +1,32 @@
// The journal row one Claude spawn group writes: its durable identity and the
// body it revises in place.
import type {
AgentJournalItemBody,
AgentJournalItemIdentity
} from '../../shared/agent-session-journal-types'
import { subagentGroupFallbackText } from '../../shared/native-chat-subagent-summary'
import type { NativeChatSubagentEntry } from '../../shared/native-chat-types'
/** Durable journal identity for the group's row — stable across revisions and
* across a restart, so replay finds the same row instead of appending a new one. */
export function claudeSubagentGroupIdentity(groupId: string): AgentJournalItemIdentity {
return { provider: 'orca', clientMessageId: `claude-subagents:${groupId}` }
}
/** The roster row: the structured block plus the plain sentence an older client
* renders in its place. A message whose only block is the new variant would
* reach such a client with nothing it can draw. */
export function claudeSubagentGroupBody(
groupId: string,
agents: readonly NativeChatSubagentEntry[]
): AgentJournalItemBody {
return {
kind: 'message',
role: 'system',
blocks: [
{ type: 'text', text: subagentGroupFallbackText(agents) },
{ type: 'subagent-group', groupId, agents: [...agents] }
]
}
}
@@ -0,0 +1,60 @@
import { describe, expect, it } from 'vitest'
import { ClaudeSubagentIds } from './claude-subagent-id-aliases'
describe('ClaudeSubagentIds', () => {
it('resolves an aliased tool id to its task, and an unaliased id to itself', () => {
const ids = new ClaudeSubagentIds()
ids.alias('toolu_1', 'task-1')
expect(ids.canonical('toolu_1')).toBe('task-1')
expect(ids.canonical('toolu_unknown')).toBe('toolu_unknown')
})
it('remembers an exclusion under either of the ids that named it', () => {
const ids = new ClaudeSubagentIds()
ids.exclude('task-bash')
expect(ids.isExcluded('toolu_bash', 'task-bash')).toBe(true)
expect(ids.isExcluded(null, null)).toBe(false)
expect(ids.isExcluded('task-agent')).toBe(false)
})
it('drops the oldest alias past the bound and keeps the newest', () => {
const ids = new ClaudeSubagentIds()
for (let index = 0; index <= 512; index += 1) {
ids.alias(`toolu_${index}`, `task-${index}`)
}
// Evicted: the id now stands only for itself.
expect(ids.canonical('toolu_0')).toBe('toolu_0')
expect(ids.canonical('toolu_512')).toBe('task-512')
expect(ids.canonical('toolu_1')).toBe('task-1')
})
it('drops the oldest exclusion past the bound and keeps the newest', () => {
const ids = new ClaudeSubagentIds()
for (let index = 0; index <= 512; index += 1) {
ids.exclude(`task-${index}`)
}
expect(ids.isExcluded('task-0')).toBe(false)
expect(ids.isExcluded('task-512')).toBe(true)
expect(ids.isExcluded('task-1')).toBe(true)
})
it('does not retain oversized aliases or exclusions', () => {
const ids = new ClaudeSubagentIds()
const oversized = 'x'.repeat(513)
ids.alias(oversized, 'task-1')
ids.alias('tool-1', oversized)
ids.exclude(oversized)
expect(ids.canonical(oversized)).toBe(oversized)
expect(ids.canonical('tool-1')).toBe('tool-1')
expect(ids.isExcluded(oversized)).toBe(false)
})
it('forgets everything on clear', () => {
const ids = new ClaudeSubagentIds()
ids.alias('toolu_1', 'task-1')
ids.exclude('task-1')
ids.clear()
expect(ids.canonical('toolu_1')).toBe('toolu_1')
expect(ids.isExcluded('task-1')).toBe(false)
})
})
@@ -0,0 +1,63 @@
// Which Claude ids name the same subagent, and which name no subagent at all.
//
// Claude re-announces a resumed task under a NEW `tool_use_id` while `task_id`
// stays put, so tool ids are aliases of a canonical task id — a store keyed on
// the tool id would show the child twice after every resume.
//
// The exclusions matter just as much: `task_updated` carries no `task_type` and
// child traffic carries no task metadata at all, so the one announcement that
// said "this is a backgrounded shell, not an agent" has to be remembered or a
// later frame re-admits it.
import { isBoundedClaudeTaskId } from './claude-background-task-tracker'
/** Both maps are event-accumulated and nothing prunes them, so both are bounded. */
const MAX_TOOL_USE_ALIASES = 512
const MAX_EXCLUDED_IDS = 512
export class ClaudeSubagentIds {
private readonly canonicalByToolUse = new Map<string, string>()
private readonly excluded = new Set<string>()
/** The task id a tool id stands for, or the id itself when nothing aliases it. */
canonical(id: string): string {
return this.canonicalByToolUse.get(id) ?? id
}
alias(toolUseId: string, taskId: string): void {
if (!isBoundedClaudeTaskId(toolUseId) || !isBoundedClaudeTaskId(taskId)) {
return
}
this.canonicalByToolUse.set(toolUseId, taskId)
while (this.canonicalByToolUse.size > MAX_TOOL_USE_ALIASES) {
const oldest = this.canonicalByToolUse.keys().next()
if (oldest.done || oldest.value === toolUseId) {
break
}
this.canonicalByToolUse.delete(oldest.value)
}
}
exclude(id: string): void {
if (!isBoundedClaudeTaskId(id)) {
return
}
this.excluded.add(id)
while (this.excluded.size > MAX_EXCLUDED_IDS) {
const oldest = this.excluded.values().next()
if (oldest.done || oldest.value === id) {
break
}
this.excluded.delete(oldest.value)
}
}
isExcluded(...ids: (string | null)[]): boolean {
return ids.some((id) => id !== null && this.excluded.has(id))
}
clear(): void {
this.canonicalByToolUse.clear()
this.excluded.clear()
}
}
@@ -0,0 +1,75 @@
import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types'
import type { NativeChatSubagentEntry } from '../../shared/native-chat-types'
import type { ClaudeSubagentTaskFrame } from './claude-subagent-task-frames'
const MAX_INVOCATIONS_PER_SUBAGENT = 16
export type TrackedEntry = {
entry: NativeChatSubagentEntry
/** The only signal separating a child that dies with its turn from one told to
* outlive it. A turn-end sweep must leave a backgrounded child alone. */
backgrounded: boolean
toolUseId: string | null
invocationIds: Set<string> | null
/** Label before its ordinal suffix, so a later announcement can tell a
* provisional row from one that already carries the provider's own name. */
labelBase: string
}
export type RosterGroup = {
groupId: string
identity: AgentJournalItemIdentity
/** Insertion order is the display order; the map holds the state. */
entries: Map<string, TrackedEntry>
/** Lifetime admissions bound retained labels even when entries are removed. */
admittedEntries: number
/** Labels remain reserved after removal or provisional-name replacement. */
claimedLabels: Set<string>
/** Last body written, so an idempotent replay writes no new revision. */
lastSerialized: string | null
}
// Invocation history stays with the entry, independent of the evicting alias cache.
export function applyClaudeSubagentInvocation(
tracked: TrackedEntry,
frame: ClaudeSubagentTaskFrame,
now: () => number
): boolean {
if (tracked.invocationIds === null) {
return false
}
const newInvocation =
frame.announcement && frame.toolUseId !== null && !tracked.invocationIds.has(frame.toolUseId)
if (newInvocation && frame.toolUseId) {
if (tracked.invocationIds.size >= MAX_INVOCATIONS_PER_SUBAGENT) {
tracked.invocationIds = null
tracked.entry = { ...tracked.entry, state: 'unverifiable', settledAt: now() }
return true
}
tracked.invocationIds.add(frame.toolUseId)
if (tracked.toolUseId !== null && tracked.toolUseId !== frame.toolUseId) {
tracked.backgrounded = frame.backgrounded ?? false
tracked.entry = { ...tracked.entry, state: frame.state ?? 'working', settledAt: undefined }
}
tracked.toolUseId = frame.toolUseId
} else if (tracked.toolUseId && frame.toolUseId && tracked.toolUseId !== frame.toolUseId) {
return false
}
if (tracked.toolUseId === null) {
tracked.toolUseId = frame.toolUseId
}
return true
}
/** Two children can share a description; the ordinal keeps their rows apart
* without inventing a name the provider never sent. The probe is over the
* labels actually rendered, not a per-base counter: a generated `Audit 2`
* must not collide with a provider that names its own child `Audit 2`. */
export function claimClaudeSubagentLabel(group: RosterGroup, base: string): string {
let candidate = base
for (let ordinal = 2; group.claimedLabels.has(candidate); ordinal++) {
candidate = `${base} ${ordinal}`
}
group.claimedLabels.add(candidate)
return candidate
}
@@ -0,0 +1,602 @@
import { describe, expect, it, vi } from 'vitest'
import type {
AgentJournalItemBody,
AgentJournalItemIdentity
} from '../../shared/agent-session-journal-types'
import type {
NativeChatSubagentEntry,
NativeChatSubagentGroupBlock
} from '../../shared/native-chat-types'
import type { AgentSessionJournal } from '../native-chat/agent-session-journal/journal-store'
import {
createDeferredStructuredAgentSessionEventSink,
type StructuredAgentSessionEventSink
} from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
import { ClaudeSubagentRoster } from './claude-subagent-roster'
const TURN_1 = 'claude-session:turn-1'
function agentsOf(body: AgentJournalItemBody | undefined): NativeChatSubagentEntry[] {
if (!body || body.kind !== 'message') {
return []
}
const block = body.blocks.find(
(candidate): candidate is NativeChatSubagentGroupBlock => candidate.type === 'subagent-group'
)
return block ? block.agents : []
}
function isGroupRow(identity: AgentJournalItemIdentity, groupId: string): boolean {
return identity.provider === 'orca' && identity.clientMessageId === `claude-subagents:${groupId}`
}
function harness(groupKey: string | null = TURN_1) {
const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = []
const tombstones: AgentJournalItemIdentity[] = []
const sink: StructuredAgentSessionEventSink = {
appendItem: (identity, body) => items.push({ identity, body }),
appendTombstone: (identity) => tombstones.push(identity),
publish: vi.fn()
}
let clock = 1_000
let key = groupKey
const roster = new ClaudeSubagentRoster({
sink,
currentGroupKey: () => key,
now: () => (clock += 1)
})
const roles = (): NativeChatSubagentEntry[] => agentsOf(items.at(-1)?.body)
/** The last row written for one group, so a test can read a row that is no
* longer the newest one. */
const rolesIn = (groupId: string): NativeChatSubagentEntry[] =>
agentsOf(items.findLast((item) => isGroupRow(item.identity, groupId))?.body)
return {
roster,
items,
tombstones,
roles,
rolesIn,
setGroupKey: (next: string | null) => {
key = next
}
}
}
function system(subtype: string, fields: Record<string, unknown>): Record<string, unknown> {
return { type: 'system', subtype, session_id: 'claude-session', ...fields }
}
function started(fields: Record<string, unknown>): Record<string, unknown> {
return system('task_started', { task_type: 'local_agent', ...fields })
}
describe('ClaudeSubagentRoster', () => {
it('builds the row from task_started, with the fallback sentence beside the block', () => {
const { roster, items, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Review the diff' })
)
expect(items).toHaveLength(1)
expect(items[0]?.identity).toEqual({
provider: 'orca',
clientMessageId: 'claude-subagents:claude-session:turn-1'
})
const body = items[0]?.body
expect(body?.kind === 'message' && body.blocks[0]).toEqual({
type: 'text',
text: 'Kicked off 1 subagent'
})
expect(roles()).toEqual([
expect.objectContaining({ id: 'task-1', label: 'Review the diff', state: 'working' })
])
})
it('keeps a backgrounded shell task out of the roster', () => {
const { roster, items } = harness()
roster.observeSystemFrame(
system('task_started', {
task_id: 'task-bash',
tool_use_id: 'toolu_bash',
task_type: 'local_bash',
description: 'sleep 20',
is_backgrounded: true
})
)
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-bash', patch: { status: 'running' } })
)
// Its own frames carry a tool_use_id, so only the excluded-id memory stops it.
roster.observeChildActivity('toolu_bash')
expect(items).toHaveLength(0)
})
it('never renders a task marked skip_transcript', () => {
const { roster, items } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-a', tool_use_id: 'toolu_a', skip_transcript: true })
)
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-a', patch: { status: 'completed' } })
)
roster.observeChildActivity('toolu_a')
expect(items).toHaveLength(0)
})
it('drops a provisional row once an announcement says the task is not a subagent', () => {
const { roster, items, tombstones, roles } = harness()
roster.observeChildActivity('toolu_bash')
expect(roles()).toHaveLength(1)
roster.observeSystemFrame(
system('task_started', {
task_id: 'task-bash',
tool_use_id: 'toolu_bash',
task_type: 'local_bash'
})
)
expect(tombstones).toEqual([
{ provider: 'orca', clientMessageId: 'claude-subagents:claude-session:turn-1' }
])
expect(items).toHaveLength(1)
})
it('does not duplicate a resumed task re-announced under a new tool_use_id', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_first', description: 'Audit' })
)
roster.observeChildActivity('toolu_first')
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_second', description: 'Audit' })
)
roster.observeChildActivity('toolu_second')
expect(roles()).toEqual([
expect.objectContaining({ id: 'task-1', label: 'Audit', state: 'working' })
])
})
it('adopts a row built from child traffic when the announcement finally names it', () => {
const { roster, roles } = harness()
roster.observeChildActivity('toolu_1')
expect(roles()).toEqual([expect.objectContaining({ id: 'toolu_1', label: 'subagent' })])
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Explore' })
)
expect(roles()).toEqual([
expect.objectContaining({ id: 'task-1', label: 'Explore', state: 'working' })
])
})
it('is idempotent: a repeated frame writes no new revision', () => {
const { roster, items } = harness()
const frame = started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Audit' })
roster.observeSystemFrame(frame)
roster.observeSystemFrame(frame)
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { status: 'running' } })
)
expect(items).toHaveLength(1)
})
it('latches a terminal state against a later live report', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' }))
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { status: 'failed' } })
)
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { status: 'running' } })
)
expect(roles()).toEqual([expect.objectContaining({ state: 'failed' })])
})
it('ignores an update for a task it never rostered', () => {
const { roster, items } = harness()
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-unknown', patch: { status: 'running' } })
)
expect(items).toHaveLength(0)
})
it('disambiguates children that share a description', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Explore' }))
roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Explore' }))
expect(roles().map((agent) => agent.label)).toEqual(['Explore', 'Explore 2'])
})
describe('turn end', () => {
it('leaves a backgrounded child working and marks a foreground one unverifiable', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-fg', description: 'Foreground' }))
roster.observeSystemFrame(
started({ task_id: 'task-bg', description: 'Background', is_backgrounded: true })
)
roster.settleTurn(TURN_1)
expect(roles()).toEqual([
expect.objectContaining({ label: 'Foreground', state: 'unverifiable' }),
expect.objectContaining({ label: 'Background', state: 'working' })
])
})
it('never re-settles a child that already reported an outcome', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' }))
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { status: 'completed' } })
)
roster.settleTurn(TURN_1)
expect(roles()).toEqual([expect.objectContaining({ state: 'completed' })])
})
it('sweeps backgrounded children only when the provider itself is gone', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-bg', description: 'Background', is_backgrounded: true })
)
roster.settleTurn(TURN_1)
roster.settleSession()
expect(roles()).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
})
describe('spawn tool result', () => {
it('settles a foreground child', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'toolu_1' }))
roster.observeToolResult('toolu_1', false)
expect(roles()).toEqual([expect.objectContaining({ state: 'completed' })])
})
it('reports a failed spawn as failed', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'toolu_1' }))
roster.observeToolResult('toolu_1', true)
expect(roles()).toEqual([expect.objectContaining({ state: 'failed' })])
})
it('ignores the immediate result a backgrounded spawn returns', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_1', is_backgrounded: true })
)
roster.observeToolResult('toolu_1', false)
expect(roles()).toEqual([expect.objectContaining({ state: 'working' })])
})
it('ignores results for tools that are not spawn calls', () => {
const { roster, items } = harness()
roster.observeToolResult('toolu_read', false)
expect(items).toHaveLength(0)
})
})
describe('label ordinals', () => {
it('never re-issues an ordinal a removed row gave up', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' }))
roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Audit' }))
// task-1 is re-announced as a shell task, so its row goes; reclaiming the
// ordinal it held would print a second 'Audit 2' beside the one still shown.
roster.observeSystemFrame(
system('task_started', { task_id: 'task-1', task_type: 'local_bash' })
)
roster.observeSystemFrame(started({ task_id: 'task-3', description: 'Audit' }))
expect(roles().map((agent) => agent.label)).toEqual(['Audit 2', 'Audit 3'])
})
it('never generates a label a provider-supplied one already took', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' }))
roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Audit' }))
// The provider's own name for the third child is the label the ordinal just
// generated for the second; a per-base counter would print it twice.
roster.observeSystemFrame(started({ task_id: 'task-3', description: 'Audit 2' }))
const labels = roles().map((agent) => agent.label)
expect(labels).toEqual(['Audit', 'Audit 2', 'Audit 2 2'])
expect(new Set(labels).size).toBe(labels.length)
})
})
describe('child traffic for an id the CLI never declared', () => {
it('creates nothing once the CLI has announced any task at all', () => {
const { roster, items } = harness()
// A rejected announcement still proves this CLI declares what it spawns.
roster.observeSystemFrame(
system('task_started', { task_id: 'task-bash', task_type: 'local_bash' })
)
roster.observeChildActivity('toolu_never_announced')
expect(items).toHaveLength(0)
})
it('rejects an over-long provisional id instead of storing it as an entry id', () => {
const { roster, items } = harness()
// The announced path drops an id past `claudeTaskId`'s bound; the
// provisional one writes the same durable entry id, so it must too.
roster.observeChildActivity(`toolu_${'x'.repeat(512)}`)
expect(items).toHaveLength(0)
roster.observeChildActivity(`toolu_${'x'.repeat(500)}`)
expect(items).toHaveLength(1)
})
it('still rosters a subagent announced after a task the filter rejected', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
system('task_started', { task_id: 'task-bash', task_type: 'local_bash' })
)
// The gate closes the child-traffic fallback, never the announcement path.
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Explore' })
)
roster.observeChildActivity('toolu_1')
expect(roles()).toEqual([
expect.objectContaining({ id: 'task-1', label: 'Explore', state: 'working' })
])
})
it('leaves a grandchild parented inside the sidechain out of the roster', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Explore' })
)
roster.observeChildActivity('toolu_1')
// A tool the subagent itself ran: never announced, so never excluded either.
roster.observeChildActivity('toolu_inner')
expect(roles()).toEqual([
expect.objectContaining({ id: 'task-1', label: 'Explore', state: 'working' })
])
})
it('still mints the provisional row for a release that announces no task', () => {
const { roster, roles } = harness()
roster.observeChildActivity('toolu_1')
// Not an announcement: the fallback path stays open for this release.
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-x', patch: { status: 'running' } })
)
roster.observeChildActivity('toolu_2')
expect(roles().map((agent) => agent.label)).toEqual(['subagent', 'subagent 2'])
})
})
describe('groups that no later event can reach', () => {
it('loses contact with a group evicted past the bound', () => {
const { roster, rolesIn, setGroupKey } = harness('turn-0')
for (let index = 0; index < 33; index += 1) {
setGroupKey(`turn-${index}`)
roster.observeSystemFrame(started({ task_id: `task-${index}`, description: 'Audit' }))
}
expect(rolesIn('turn-0')).toEqual([expect.objectContaining({ state: 'unverifiable' })])
expect(rolesIn('turn-32')).toEqual([expect.objectContaining({ state: 'working' })])
})
it('loses contact with a live child when the translator is disposed without an end', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-bg', description: 'Background', is_backgrounded: true })
)
roster.dispose()
expect(roles()).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
it('writes nothing on dispose when the session already settled', () => {
const { roster, items } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' }))
roster.settleSession()
const written = items.length
roster.dispose()
expect(items).toHaveLength(written)
})
})
it('groups children outside any turn under their own row', () => {
const { roster, items } = harness(null)
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' }))
expect(items[0]?.identity).toEqual({
provider: 'orca',
clientMessageId: 'claude-subagents:outside-turn'
})
})
})
describe('ClaudeSubagentRoster — the turn that is ending', () => {
it('leaves a child announced outside any turn alone when an unrelated turn ends', () => {
const { roster, rolesIn, setGroupKey } = harness(null)
roster.observeSystemFrame(started({ task_id: 'task-early', description: 'Early' }))
setGroupKey(TURN_1)
roster.observeSystemFrame(started({ task_id: 'task-turn', description: 'In turn' }))
roster.settleTurn(TURN_1)
expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'working' })])
expect(rolesIn(TURN_1)).toEqual([expect.objectContaining({ state: 'unverifiable' })])
// `unverifiable` latches, so sweeping it above would have swallowed this.
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-early', patch: { status: 'completed' } })
)
expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'completed' })])
})
it('sweeps the outside-turn group when a turn with no key of its own ends', () => {
const { roster, rolesIn } = harness(null)
roster.observeSystemFrame(started({ task_id: 'task-early', description: 'Early' }))
roster.settleTurn(null)
expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
it('still settles an outside-turn child once the session itself ends', () => {
const { roster, rolesIn, setGroupKey } = harness(null)
roster.observeSystemFrame(started({ task_id: 'task-early', description: 'Early' }))
setGroupKey(TURN_1)
roster.observeSystemFrame(started({ task_id: 'task-turn', description: 'In turn' }))
roster.settleTurn(TURN_1)
roster.settleSession()
expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'unverifiable' })])
})
it('sweeps the turn that ended, not whichever turn is live now', () => {
const { roster, rolesIn, setGroupKey } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'First turn' }))
setGroupKey('claude-session:turn-2')
roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Second turn' }))
// Turn 1's result lands after turn 2 has already begun.
roster.settleTurn(TURN_1)
expect(rolesIn(TURN_1)).toEqual([expect.objectContaining({ state: 'unverifiable' })])
expect(rolesIn('claude-session:turn-2')).toEqual([
expect.objectContaining({ state: 'working' })
])
})
})
describe('ClaudeSubagentRoster — through the real sink queue', () => {
it('lands every revision, not just the one that was already in flight', async () => {
const appended: AgentJournalItemBody[] = []
let published = 0
const journal = {
appendItem: async (_identity: AgentJournalItemIdentity, body: AgentJournalItemBody) => {
appended.push(body)
return { cursor: { epoch: 'e', sequence: appended.length } }
},
appendTombstone: async () => ({ epoch: 'e', sequence: 0 })
} as unknown as AgentSessionJournal
const deferred = createDeferredStructuredAgentSessionEventSink()
deferred.bind({
journal,
fence: 1,
publish: () => {
published += 1
}
})
const roster = new ClaudeSubagentRoster({ sink: deferred.sink, currentGroupKey: () => TURN_1 })
// The first append is in flight while the rest are submitted, so a publish
// sharing the row's coalescing key would evict them.
roster.observeSystemFrame(started({ task_id: 'task-1', description: 'One' }))
roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Two' }))
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { status: 'completed' } })
)
const drained = await deferred.drained()
expect(drained).toEqual({ ok: true })
expect(agentsOf(appended.at(-1))).toEqual([
expect.objectContaining({ id: 'task-1', label: 'One', state: 'completed' }),
expect.objectContaining({ id: 'task-2', label: 'Two', state: 'working' })
])
expect(published).toBeGreaterThan(0)
})
})
describe('ClaudeSubagentRoster — authoritative outcomes and retained budgets', () => {
it('accepts a notification after the foreground turn lost contact', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1' }))
roster.settleTurn(TURN_1)
roster.observeSystemFrame(
system('task_notification', { task_id: 'task-1', status: 'completed' })
)
expect(roles()).toEqual([expect.objectContaining({ state: 'completed' })])
})
it('settles a background child from its notification without a task_updated', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', is_backgrounded: true }))
roster.settleTurn(TURN_1)
roster.observeSystemFrame(system('task_notification', { task_id: 'task-1', status: 'failed' }))
expect(roles()).toEqual([expect.objectContaining({ state: 'failed' })])
})
it('bounds lifetime admissions when reclassification repeatedly removes entries', () => {
const { roster, items } = harness()
for (let i = 0; i < 100; i++) {
roster.observeSystemFrame(started({ task_id: `task-${i}`, description: `Agent ${i}` }))
roster.observeSystemFrame(
system('task_started', { task_id: `task-${i}`, task_type: 'local_bash' })
)
}
expect(items).toHaveLength(64)
})
})
describe('ClaudeSubagentRoster — resumed invocation', () => {
it('reopens one canonical child on a new announcement without replaying old results', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' }))
roster.observeToolResult('first', false)
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'resumed', is_backgrounded: true })
)
expect(roles()).toEqual([expect.objectContaining({ id: 'task-1', state: 'working' })])
expect(roles()[0].settledAt).toBeUndefined()
roster.observeSystemFrame(
system('task_notification', { task_id: 'task-1', tool_use_id: 'first', status: 'completed' })
)
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' }))
expect(roles()[0].state).toBe('working')
roster.observeSystemFrame(
system('task_notification', {
task_id: 'task-1',
tool_use_id: 'resumed',
status: 'completed'
})
)
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'resumed', is_backgrounded: true })
)
expect(roles()[0].state).toBe('completed')
})
})
describe('ClaudeSubagentRoster — invocation fences', () => {
it('ignores a previous invocation tool result even without a background flag', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' }))
roster.observeToolResult('first', false)
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'next' }))
roster.observeToolResult('first', true)
expect(roles()[0].state).toBe('working')
roster.observeToolResult('next', false)
expect(roles()[0].state).toBe('completed')
})
it('does not treat an evicted alias as a new invocation', () => {
const { roster, rolesIn, setGroupKey } = harness()
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' }))
roster.observeToolResult('first', false)
setGroupKey('churn')
for (let i = 0; i < 513; i++) {
roster.observeSystemFrame(
system('task_updated', { task_id: `other-${i}`, tool_use_id: `tool-${i}` })
)
}
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' }))
expect(rolesIn(TURN_1)[0].state).toBe('completed')
})
it('bounds invocation history and refuses to reopen beyond the retained budget', () => {
const { roster, roles } = harness()
for (let i = 0; i < 20; i++) {
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: `tool-${i}` }))
if (i >= 16) {
expect(roles()[0].state).toBe('unverifiable')
}
roster.observeToolResult(`tool-${i}`, false)
}
roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'tool-0' }))
expect(roles()[0].state).toBe('unverifiable')
})
})
it('merges an explicit foreground patch without clearing on absent metadata', () => {
const { roster, roles } = harness()
roster.observeSystemFrame(
started({ task_id: 'task-1', tool_use_id: 'tool', is_backgrounded: true })
)
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { description: 'Audit' } })
)
roster.observeToolResult('tool', false)
expect(roles()[0].state).toBe('working')
roster.observeSystemFrame(
system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: false } })
)
roster.observeToolResult('tool', false)
expect(roles()[0].state).toBe('completed')
})
+388
View File
@@ -0,0 +1,388 @@
// The Claude subagent roster: one journal row per turn that spawned children.
//
// Entries are built from `task_started`, never from child traffic: a
// BACKGROUNDED subagent emits no child frames at all, so a roster fed by
// `parent_tool_use_id` alone would leave every one of them an unlabelled row
// forever. Child traffic only creates an entry for CLI releases that announce
// no task frames.
//
// Claude re-announces a resumed task under a NEW `tool_use_id`, so `task_id` is
// the key and tool ids are aliases; keying on the tool id would duplicate the
// child on every resume. Outcomes latch within an invocation; a new spawn
// alias can reopen it, and authoritative evidence can correct lost contact.
import {
canReplaceSubagentState,
isTerminalSubagentState
} from '../../shared/native-chat-subagent-summary'
import type { NativeChatSubagentEntry } from '../../shared/native-chat-types'
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
import { isBoundedClaudeTaskId } from './claude-background-task-tracker'
import { claudeSubagentGroupBody, claudeSubagentGroupIdentity } from './claude-subagent-group-row'
import { ClaudeSubagentIds } from './claude-subagent-id-aliases'
import { readClaudeSubagentTaskFrame } from './claude-subagent-task-frames'
import {
applyClaudeSubagentInvocation,
claimClaudeSubagentLabel,
type RosterGroup,
type TrackedEntry
} from './claude-subagent-roster-state'
/** Spawn-group rows kept live per session, and children per row. Both bound an
* event-accumulated map that no provider snapshot ever prunes. */
const MAX_SUBAGENT_GROUPS = 32
const MAX_SUBAGENTS_PER_GROUP = 64
/** The turn a group belongs to when Claude reports a task outside any turn. */
const OUTSIDE_TURN = 'outside-turn'
const UNLABELLED_AGENT = 'subagent'
export type ClaudeSubagentRosterDeps = {
sink: StructuredAgentSessionEventSink
/** The turn that owns children spawned right now; null outside any turn. */
currentGroupKey: () => string | null
now?: () => number
}
export class ClaudeSubagentRoster {
private readonly groups = new Map<string, RosterGroup>()
/** Canonical id → the group holding its entry, so a late update for a child
* from an earlier turn revises that turn's row instead of the live one. */
private readonly groupIdByEntry = new Map<string, string>()
private readonly ids = new ClaudeSubagentIds()
/** Set by ANY `task_started`, including one the subagent filter rejects. Once
* this CLI has proven it declares its tasks, child traffic for an id it never
* announced is a nested tool or a grandchild, not a subagent. */
private announcesTasks = false
private readonly now: () => number
constructor(private readonly deps: ClaudeSubagentRosterDeps) {
this.now = deps.now ?? (() => Date.now())
}
/** Consume a `message:system:task_*` frame. Returns false when it is not one. */
observeSystemFrame(message: Record<string, unknown>): boolean {
const frame = readClaudeSubagentTaskFrame(message)
if (!frame) {
return false
}
this.announcesTasks ||= frame.announcement
if (frame.excluded) {
// Child traffic may already have built a provisional row under the tool id;
// the announcement is the first frame that says it is not a subagent.
for (const id of [frame.taskId, frame.toolUseId]) {
if (id !== null) {
this.ids.exclude(id)
this.remove(id)
}
}
return true
}
if (this.ids.isExcluded(frame.taskId, frame.toolUseId)) {
return true
}
if (frame.toolUseId) {
this.ids.alias(frame.toolUseId, frame.taskId)
}
const located =
this.locate(frame.taskId) ??
(frame.toolUseId ? this.adopt(frame.toolUseId, frame.taskId) : null)
if (!located) {
if (frame.announcesSubagent) {
this.create(
frame.taskId,
frame.label,
frame.state ?? 'working',
frame.backgrounded ?? false,
frame.toolUseId
)
}
return true
}
const tracked = located.group.entries.get(frame.taskId)
if (tracked && !applyClaudeSubagentInvocation(tracked, frame, this.now)) {
return true
}
this.revise(located.group, frame.taskId, {
label: frame.label,
state: frame.state,
backgrounded: frame.backgrounded
})
return true
}
/**
* A frame carrying `parent_tool_use_id` — the child's own traffic. It refreshes
* nothing on an announced child; it exists so a CLI release that sends no task
* frames still shows the subagent it is running.
*/
observeChildActivity(parentToolUseId: string): void {
const canonical = this.ids.canonical(parentToolUseId)
if (this.ids.isExcluded(parentToolUseId, canonical)) {
return
}
if (this.locate(canonical)) {
return
}
if (this.announcesTasks) {
// A nested Task, a workflow child, or a grandchild parented to a tool id
// inside the sidechain all reach here. This CLI announces what it spawns,
// so an id it never declared cannot be a subagent — and a row invented for
// one is unlabelled forever and can only ever end `unverifiable`. The
// bounded exclusion set cannot cover an id that was never announced.
return
}
if (!isBoundedClaudeTaskId(canonical)) {
// `claudeTaskId` rejects an over-long announced id rather than truncating
// it; a provisional id becomes the same durable entry key, so it cannot
// enter under a looser rule.
return
}
this.create(canonical, null, 'working', false, parentToolUseId)
}
/**
* The parent turn's tool result for a spawn call. It settles a foreground
* child, whose result IS the turn's evidence the child finished. A backgrounded
* child's spawn call returns immediately while the child keeps running, so its
* result proves nothing and is ignored.
*/
observeToolResult(toolUseId: string, failed: boolean): void {
const canonical = this.ids.canonical(toolUseId)
const located = this.locate(canonical)
if (
!located ||
located.tracked.invocationIds === null ||
located.tracked.backgrounded ||
(located.tracked.toolUseId !== null && located.tracked.toolUseId !== toolUseId)
) {
return
}
this.revise(located.group, canonical, {
label: null,
state: failed ? 'failed' : 'completed',
backgrounded: false
})
}
/**
* The parent turn ended. A foreground child still reported as working will
* never be settled by an event, so it becomes `unverifiable`: contact was
* lost, which is NOT evidence the child exited. A backgrounded child was
* explicitly told to outlive the turn and is left alone.
*/
settleTurn(groupKey: string | null): void {
// Only the group this key names. `OUTSIDE_TURN` belongs to no turn, so an
// unrelated turn ending is no evidence about a child announced outside it.
// `settleSession` reaches what no turn does.
this.sweep(this.groups.get(groupKey ?? OUTSIDE_TURN), false)
}
/** The provider is gone. Nothing more will arrive for any child, backgrounded
* or not, so every one of them loses contact at once. */
settleSession(): void {
for (const group of this.groups.values()) {
this.sweep(group, true)
}
}
dispose(): void {
// Teardown paths reach here without an `ended` event, so a row still
// reporting `working` would have nothing left to revise it. A session that
// did settle first leaves every child terminal, so this writes nothing.
this.settleSession()
this.groups.clear()
this.groupIdByEntry.clear()
this.ids.clear()
this.announcesTasks = false
}
private sweep(group: RosterGroup | undefined, includeBackgrounded: boolean): void {
if (!group) {
return
}
let changed = false
for (const [id, tracked] of group.entries) {
if (isTerminalSubagentState(tracked.entry.state)) {
continue
}
if (tracked.backgrounded && !includeBackgrounded) {
continue
}
group.entries.set(id, {
...tracked,
entry: { ...tracked.entry, state: 'unverifiable', settledAt: this.now() }
})
changed = true
}
if (changed) {
this.write(group)
}
}
private create(
id: string,
label: string | null,
state: NativeChatSubagentEntry['state'],
backgrounded: boolean,
toolUseId: string | null
): void {
const group = this.groupFor()
if (group.admittedEntries >= MAX_SUBAGENTS_PER_GROUP) {
return
}
group.admittedEntries += 1
const now = this.now()
const labelBase = label ?? UNLABELLED_AGENT
group.entries.set(id, {
backgrounded,
toolUseId,
invocationIds: new Set(toolUseId ? [toolUseId] : []),
labelBase,
entry: {
id,
label: claimClaudeSubagentLabel(group, labelBase),
state,
startedAt: now,
...(isTerminalSubagentState(state) ? { settledAt: now } : {})
}
})
this.groupIdByEntry.set(id, group.groupId)
this.write(group)
}
private revise(
group: RosterGroup,
id: string,
change: {
label: string | null
state: NativeChatSubagentEntry['state'] | null
backgrounded: boolean | null
}
): void {
const tracked = group.entries.get(id)
if (!tracked) {
return
}
const next: TrackedEntry = {
...tracked,
backgrounded: change.backgrounded ?? tracked.backgrounded,
entry: { ...tracked.entry }
}
// A provisional row built from child traffic takes the real name the first
// announcement carries; an announced row keeps the name it was given.
if (
change.label &&
tracked.labelBase === UNLABELLED_AGENT &&
change.label !== UNLABELLED_AGENT
) {
next.labelBase = change.label
next.entry.label = claimClaudeSubagentLabel(group, change.label)
}
// Proven outcomes latch; lost contact can still receive a later verdict.
if (change.state && canReplaceSubagentState(tracked.entry.state, change.state)) {
next.entry.state = change.state
if (isTerminalSubagentState(change.state)) {
next.entry.settledAt = this.now()
}
}
group.entries.set(id, next)
this.write(group)
}
/** Re-key a provisional entry from its tool id onto the canonical task id the
* announcement finally named, so the child does not appear twice. */
private adopt(toolUseId: string, taskId: string): { group: RosterGroup } | null {
if (toolUseId === taskId) {
return null
}
const located = this.locate(toolUseId)
if (!located) {
return null
}
located.group.entries.delete(toolUseId)
located.group.entries.set(taskId, {
...located.tracked,
entry: { ...located.tracked.entry, id: taskId }
})
this.groupIdByEntry.delete(toolUseId)
this.groupIdByEntry.set(taskId, located.group.groupId)
return { group: located.group }
}
private remove(id: string): void {
const located = this.locate(id)
if (!located) {
return
}
located.group.entries.delete(id)
this.groupIdByEntry.delete(id)
this.write(located.group)
}
private locate(id: string): { group: RosterGroup; tracked: TrackedEntry } | null {
const groupId = this.groupIdByEntry.get(id)
const group = groupId === undefined ? undefined : this.groups.get(groupId)
const tracked = group?.entries.get(id)
return group && tracked ? { group, tracked } : null
}
private groupFor(): RosterGroup {
const groupId = this.deps.currentGroupKey() ?? OUTSIDE_TURN
const existing = this.groups.get(groupId)
if (existing) {
return existing
}
const group: RosterGroup = {
groupId,
identity: claudeSubagentGroupIdentity(groupId),
entries: new Map(),
admittedEntries: 0,
claimedLabels: new Set(),
lastSerialized: null
}
this.groups.set(groupId, group)
while (this.groups.size > MAX_SUBAGENT_GROUPS) {
const oldest = this.groups.keys().next()
if (oldest.done || oldest.value === groupId) {
break
}
const evicted = this.groups.get(oldest.value)
// Once the group leaves the map nothing can reach its children again —
// not even a session sweep — so contact is lost here.
this.sweep(evicted, true)
for (const id of evicted?.entries.keys() ?? []) {
this.groupIdByEntry.delete(id)
}
this.groups.delete(oldest.value)
}
return group
}
private write(group: RosterGroup): void {
const agents = [...group.entries.values()].map((tracked) => tracked.entry)
const options = { coalescingKey: `claude-subagents:${group.groupId}` }
if (agents.length === 0) {
// The row's last child turned out not to be a subagent. An empty roster is
// not a roster of nothing, so the row goes rather than reading "Ran 0".
if (group.lastSerialized !== null) {
group.lastSerialized = null
this.deps.sink.appendTombstone(group.identity, options)
this.deps.sink.publish()
}
return
}
const body = claudeSubagentGroupBody(group.groupId, agents)
const serialized = JSON.stringify(body)
if (serialized === group.lastSerialized) {
// Nothing changed — a duplicate delivery must not burn a revision.
return
}
group.lastSerialized = serialized
this.deps.sink.appendItem(group.identity, body, options)
// Publish keeps the sink's own coalescing slot: sharing the row's key makes
// each queued publish evict the append it was meant to flush.
this.deps.sink.publish()
}
}
@@ -0,0 +1,201 @@
import { describe, expect, it } from 'vitest'
import { readClaudeSubagentTaskFrame } from './claude-subagent-task-frames'
function system(subtype: string, fields: Record<string, unknown>): Record<string, unknown> {
return { type: 'system', subtype, session_id: 'claude-session', ...fields }
}
describe('readClaudeSubagentTaskFrame', () => {
it('ignores frames that are not task frames', () => {
expect(readClaudeSubagentTaskFrame({ type: 'assistant', subtype: 'task_started' })).toBeNull()
expect(readClaudeSubagentTaskFrame(system('init', { task_id: 'task-1' }))).toBeNull()
expect(readClaudeSubagentTaskFrame(system('task_started', {}))).toBeNull()
expect(readClaudeSubagentTaskFrame(system('task_started', { task_id: '' }))).toBeNull()
})
describe('task_type triage', () => {
it('announces a local_agent task', () => {
const frame = readClaudeSubagentTaskFrame(
system('task_started', {
task_id: 'task-1',
tool_use_id: 'toolu_1',
task_type: 'local_agent',
subagent_type: 'code-reviewer',
description: 'Review the diff'
})
)
expect(frame).toMatchObject({
taskId: 'task-1',
toolUseId: 'toolu_1',
label: 'Review the diff',
announcesSubagent: true,
excluded: false
})
})
it('excludes a backgrounded shell command even though it carries a tool_use_id', () => {
const frame = readClaudeSubagentTaskFrame(
system('task_started', {
task_id: 'task-bash',
tool_use_id: 'toolu_bash',
task_type: 'local_bash',
description: 'sleep 20',
is_backgrounded: true
})
)
expect(frame).toMatchObject({
taskId: 'task-bash',
toolUseId: 'toolu_bash',
announcesSubagent: false,
excluded: true
})
})
it('excludes workflows and monitors', () => {
for (const taskType of ['local_workflow', 'monitor']) {
expect(
readClaudeSubagentTaskFrame(
system('task_started', { task_id: `task-${taskType}`, task_type: taskType })
)
).toMatchObject({ announcesSubagent: false, excluded: true })
}
})
it('caps a subagent_type label the way a description is capped', () => {
const frame = readClaudeSubagentTaskFrame(
system('task_started', { task_id: 'task-1', subagent_type: 'a'.repeat(900) })
)
// The roster stores this label verbatim, so nothing downstream bounds it.
expect(frame?.label).toHaveLength(512)
})
it('falls back to subagent_type only when the release sends no task_type', () => {
expect(
readClaudeSubagentTaskFrame(
system('task_started', { task_id: 'task-old', subagent_type: 'explorer' })
)
).toMatchObject({ announcesSubagent: true, label: 'explorer' })
expect(
readClaudeSubagentTaskFrame(system('task_started', { task_id: 'task-bare' }))
).toMatchObject({ announcesSubagent: false, excluded: true })
// A type this build does not recognise is not an agent on subagent_type's word.
expect(
readClaudeSubagentTaskFrame(
system('task_started', {
task_id: 'task-new',
task_type: 'local_something_new',
subagent_type: 'explorer'
})
)
).toMatchObject({ announcesSubagent: false, excluded: true })
})
it('excludes ambient housekeeping tasks', () => {
for (const suppression of [{ skip_transcript: true }, { ambient: true }]) {
expect(
readClaudeSubagentTaskFrame(
system('task_started', {
task_id: 'task-ambient',
task_type: 'local_agent',
subagent_type: 'watcher',
...suppression
})
)
).toMatchObject({ announcesSubagent: false, excluded: true })
}
})
})
describe('status', () => {
it('collapses every in-flight status to working', () => {
for (const status of ['pending', 'running', 'paused']) {
expect(
readClaudeSubagentTaskFrame(
system('task_updated', { task_id: 'task-1', patch: { status } })
)
).toMatchObject({ state: 'working' })
}
})
it('maps the settled statuses onto the carrier vocabulary', () => {
const mapped: [string, string][] = [
['completed', 'completed'],
['failed', 'failed'],
['killed', 'stopped'],
['stopped', 'stopped']
]
for (const [status, state] of mapped) {
expect(
readClaudeSubagentTaskFrame(
system('task_updated', { task_id: 'task-1', patch: { status } })
)
).toMatchObject({ state })
}
})
it('reports no state for a status it cannot map', () => {
for (const status of ['__proto__', 'toString', 'invented', 7, null]) {
expect(
readClaudeSubagentTaskFrame(
system('task_updated', { task_id: 'task-1', patch: { status } })
)
).toMatchObject({ state: null })
}
})
it('treats progress as no lifecycle verdict', () => {
for (const subtype of ['task_progress']) {
expect(
readClaudeSubagentTaskFrame(
system(subtype, { task_id: 'task-1', status: 'completed', patch: { status: 'failed' } })
)
).toMatchObject({ state: null })
}
})
})
it('reads the notification verdict from its top-level status', () => {
for (const state of ['completed', 'failed', 'stopped']) {
expect(
readClaudeSubagentTaskFrame(
system('task_notification', {
task_id: 'task-1',
status: state,
patch: { status: 'running' }
})
)
).toMatchObject({ state })
}
})
it('reads the backgrounded flag from the frame or its patch', () => {
expect(
readClaudeSubagentTaskFrame(
system('task_started', {
task_id: 'task-1',
task_type: 'local_agent',
is_backgrounded: true
})
)
).toMatchObject({ backgrounded: true })
expect(
readClaudeSubagentTaskFrame(
system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: true } })
)
).toMatchObject({ backgrounded: true })
expect(
readClaudeSubagentTaskFrame(system('task_updated', { task_id: 'task-1', patch: {} }))
).toMatchObject({ backgrounded: null })
})
it('collapses a multi-line description into one bounded label', () => {
expect(
readClaudeSubagentTaskFrame(
system('task_updated', {
task_id: 'task-1',
patch: { description: ' audit\n the lockfile ' }
})
)
).toMatchObject({ label: 'audit the lockfile' })
})
})
@@ -0,0 +1,123 @@
// Claude's declarative task protocol, read as subagent roster events.
//
// `local_agent`, `local_workflow` and `local_bash` tasks all arrive on the same
// `message:system:task_*` channel and ALL carry a `tool_use_id`, so id presence
// discriminates nothing: filtering on it alone puts a backgrounded `sleep 20` in
// the subagent roster. `task_type` is the discriminator, with `subagent_type`
// covering CLI releases that predate it.
import type { NativeChatSubagentState } from '../../shared/native-chat-types'
import {
classifyClaudeBackgroundTaskKind,
claudeTaskDescription,
claudeTaskId,
isBoundedClaudeTaskId
} from './claude-background-task-tracker'
import { claudeRecord, claudeText } from './claude-structured-item-translation'
const TASK_SUBTYPES: ReadonlySet<string> = new Set([
'task_started',
'task_updated',
'task_progress',
'task_notification'
])
/** Provider status → the carrier's vocabulary. `killed` and `stopped` both mean
* the task was deliberately ended, which the carrier calls `stopped`; every
* in-flight status collapses to `working`. A Map, not an object, so a payload
* carrying `__proto__` as its status cannot resolve to an inherited value. */
const TASK_STATES: ReadonlyMap<string, NativeChatSubagentState> = new Map([
['pending', 'working'],
['running', 'working'],
['paused', 'working'],
['completed', 'completed'],
['failed', 'failed'],
['killed', 'stopped'],
['stopped', 'stopped']
] satisfies [string, NativeChatSubagentState][])
export type ClaudeSubagentTaskFrame = {
/** Canonical, resume-stable id — the roster key. */
taskId: string
/** Re-minted when Claude re-announces a resumed task, so it is only an alias. */
toolUseId: string | null
label: string | null
/** null when the frame reported no lifecycle status. */
state: NativeChatSubagentState | null
backgrounded: boolean | null
/** Any `task_started`, subagent or not. Proof this CLI declares its tasks. */
announcement: boolean
/** `task_started` for a task the roster should show. Only an announcement
* creates an entry: an update carries no `task_type`, so honouring one for an
* unknown id would roster whatever else shares this channel. */
announcesSubagent: boolean
/** Ambient housekeeping, or a task that is not a subagent at all. Its ids must
* never reach the roster, by this frame or by later child traffic. */
excluded: boolean
}
/** True when the task Claude announced is a subagent rather than a backgrounded
* shell command or a workflow. */
export function isClaudeSubagentTask(message: Record<string, unknown>): boolean {
if (classifyClaudeBackgroundTaskKind(message.task_type) === 'agent') {
return true
}
// Releases predating `task_type` still name the child in `subagent_type`. A
// task_type Orca does not recognise is NOT covered: it is a type this build
// has no reason to believe is an agent.
return (
(message.task_type === undefined || message.task_type === null) &&
claudeText(message.subagent_type) !== null
)
}
function taskState(value: unknown): NativeChatSubagentState | null {
return typeof value === 'string' ? (TASK_STATES.get(value) ?? null) : null
}
export function readClaudeSubagentTaskFrame(
message: Record<string, unknown>
): ClaudeSubagentTaskFrame | null {
if (message.type !== 'system') {
return null
}
const subtype = claudeText(message.subtype)
if (!subtype || !TASK_SUBTYPES.has(subtype)) {
return null
}
const taskId = claudeTaskId(message)
if (!taskId) {
return null
}
const patch = claudeRecord(message.patch)
const toolUseId = claudeText(message.tool_use_id) ?? claudeText(patch?.tool_use_id)
const announcement = subtype === 'task_started'
// Housekeeping Claude runs for itself; the user never asked for it.
const suppressed = message.ambient === true || message.skip_transcript === true
const subagent = announcement && !suppressed && isClaudeSubagentTask(message)
return {
taskId,
toolUseId: toolUseId && isBoundedClaudeTaskId(toolUseId) ? toolUseId : null,
label:
claudeTaskDescription(message.description) ??
claudeTaskDescription(patch?.description) ??
// Bounded like a description: the roster stores whatever this returns.
(announcement ? (claudeTaskDescription(message.subagent_type) ?? null) : null),
// Notifications carry terminal evidence; progress carries usage only.
state:
subtype === 'task_notification'
? taskState(message.status)
: announcement || subtype === 'task_updated'
? taskState(patch?.status ?? message.status)
: null,
backgrounded:
typeof patch?.is_backgrounded === 'boolean'
? patch.is_backgrounded
: typeof message.is_backgrounded === 'boolean'
? message.is_backgrounded
: null,
announcement,
announcesSubagent: subagent,
excluded: announcement && !subagent
}
}
@@ -1,3 +1,4 @@
import { highestUsageKey } from '../usage/highest-usage-key'
import type {
CodexUsageBreakdownKind,
CodexUsageBreakdownRow,
@@ -57,9 +58,8 @@ export function buildSummary(
}
}
const topModel = [...byModel.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null
const topProject =
[...byProject.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null
const topModel = highestUsageKey(byModel)
const topProject = highestUsageKey(byProject)
return {
scope,
@@ -0,0 +1,128 @@
import { describe, expect, it } from 'vitest'
import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types'
import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key'
import { CodexBackgroundCommandTracker } from './codex-background-command-tracker'
import { createCodexJournalTranslator } from './codex-structured-journal-translation'
import type { CodexStructuredSessionEvent } from './codex-structured-session-state'
function notification(
method: string,
params: Record<string, unknown>
): Extract<CodexStructuredSessionEvent, { type: 'notification' }> {
return {
type: 'notification',
sessionId: 'session',
threadId: 'root',
method,
params: { threadId: 'root', turnId: 'turn', ...params }
}
}
function command(method: string, id = 'exec', threadId = 'root') {
return {
...notification(method, {
item: {
type: 'commandExecution',
id,
command: 'sleep 30',
source: 'unifiedExecStartup',
status: method === 'item/completed' ? 'completed' : 'inProgress',
exitCode: method === 'item/completed' ? 0 : null
}
}),
threadId
}
}
describe('persistent command ownership', () => {
it('preflights finite metadata capacity and admits work again after process completion', () => {
const tracker = new CodexBackgroundCommandTracker('root', 700)
const first = command('item/started', 'first')
const second = command('item/started', 'second')
expect(tracker.canObserve(first)).toBe(true)
tracker.observe(first)
expect(tracker.canObserve(second)).toBe(false)
expect(() => tracker.observe(second)).toThrow('not admitted')
expect(tracker.tasks()).toHaveLength(1)
expect(tracker.retainedMetadataBytes).toBeLessThanOrEqual(700)
tracker.observe(command('item/completed', 'first'))
expect(tracker.canObserve(second)).toBe(true)
tracker.observe(second)
expect(tracker.tasks()).toHaveLength(1)
expect(tracker.retainedMetadataBytes).toBeLessThanOrEqual(700)
tracker.clear()
expect(tracker.retainedMetadataBytes).toBe(0)
})
it('keeps the journal running across turn completion and accepts late output and exit', () => {
const rows: { key: string; body: AgentJournalItemBody }[] = []
const translator = createCodexJournalTranslator({
primaryThreadId: () => 'root',
sink: {
appendItem: (identity, body) => rows.push({ key: agentJournalItemKey(identity), body }),
appendTombstone: () => {},
publish: () => {}
}
})
const tracker = new CodexBackgroundCommandTracker('root')
const deliver = (event: Extract<CodexStructuredSessionEvent, { type: 'notification' }>) => {
expect(translator.handle(event)).toEqual({ accepted: true })
tracker.observe(event)
}
deliver(notification('turn/started', { turn: { id: 'turn' } }))
deliver(command('item/started'))
const originalKey = rows.find(({ body }) => body.kind === 'tool-call')?.key
deliver(notification('turn/completed', { turn: { id: 'turn' } }))
expect(rows.filter(({ body }) => body.kind === 'tool-call').map(({ body }) => body)).toEqual([
expect.objectContaining({ state: 'running' })
])
expect(tracker.tasks()).toHaveLength(1)
deliver(
notification('item/commandExecution/outputDelta', { itemId: 'exec', delta: 'late output' })
)
translator.flush()
expect(rows.at(-1)).toMatchObject({ key: originalKey, body: { state: 'running' } })
deliver(command('item/completed'))
expect(rows.at(-1)).toMatchObject({ key: originalKey, body: { state: 'completed' } })
expect(tracker.tasks()).toEqual([])
translator.dispose()
})
it('counts child shells only after the child stops covering them, without resurrecting exits', () => {
const tracker = new CodexBackgroundCommandTracker('root')
tracker.observe(command('item/started', 'child-exec', 'child'))
tracker.observe(
notification('item/started', {
item: {
type: 'commandExecution',
id: 'poll',
source: 'unifiedExecInteraction',
status: 'inProgress'
}
})
)
expect(tracker.tasks(new Set(['child']))).toEqual([])
expect(tracker.tasks()).toEqual([
{ id: 'codex-command:thread:child:child-exec', kind: 'command', description: 'sleep 30' }
])
tracker.observe(command('item/completed', 'child-exec', 'child'))
tracker.observe(command('item/started'))
tracker.observe(command('item/completed'))
tracker.observe(command('item/started'))
expect(tracker.tasks()).toEqual([])
})
it('retains live commands while recycling bounded settled history', () => {
const tracker = new CodexBackgroundCommandTracker('root')
tracker.observe(command('item/started', 'long-lived'))
for (let index = 0; index < 300; index += 1) {
tracker.observe(command('item/started', `short-${index}`))
tracker.observe(command('item/completed', `short-${index}`))
}
expect(tracker.tasks()).toEqual([
{ id: 'codex-command:primary:long-lived', kind: 'command', description: 'sleep 30' }
])
tracker.clear()
expect(tracker.tasks()).toEqual([])
})
})
@@ -0,0 +1,150 @@
import type { AgentSessionBackgroundTask } from '../../shared/agent-session-wire'
import type { CodexBackgroundTaskEvent } from './codex-background-task-frames'
import { codexCommandOutlivesTurn } from './codex-command-lifecycle'
import { readRecord, readString } from './codex-item-field-readers'
import { readCodexThreadItem } from './codex-structured-item-translation'
import { MAX_CODEX_ITEM_STREAM_METADATA_BYTES } from './codex-item-stream-retention'
const MAX_SETTLED_COMMANDS = 128
const MAX_DESCRIPTION_CHARS = 512
type Command = { threadId: string; task: AgentSessionBackgroundTask; bytes: number }
/** Stays within the retained bound, so read-time qualification cannot outgrow admission. */
function qualifiedDescription(label: string, description: string | undefined): string {
return (description ? `${label} — ${description}` : label).slice(0, MAX_DESCRIPTION_CHARS)
}
export class CodexBackgroundCommandTracker {
private readonly commands = new Map<string, Command>()
private readonly settled = new Map<string, number>()
private liveBytes = 0
private settledBytes = 0
constructor(
private readonly primaryThreadId: string,
private readonly maxMetadataBytes = MAX_CODEX_ITEM_STREAM_METADATA_BYTES
) {}
get retainedMetadataBytes(): number {
return this.liveBytes + this.settledBytes
}
canObserve(event: CodexBackgroundTaskEvent): boolean {
const parsed = this.parse(event)
return (
!parsed ||
parsed.completed ||
this.commands.has(parsed.key) ||
this.settled.has(parsed.key) ||
this.liveBytes + parsed.command.bytes <= this.maxMetadataBytes
)
}
observe(event: CodexBackgroundTaskEvent): void {
const parsed = this.parse(event)
if (!parsed || this.settled.has(parsed.key)) {
return
}
const { key, command, completed } = parsed
const existing = this.commands.get(key)
if (completed) {
if (existing) {
this.liveBytes -= existing.bytes
this.commands.delete(key)
}
const bytes = Buffer.byteLength(key, 'utf8') + 256
if (this.liveBytes + bytes <= this.maxMetadataBytes) {
this.settled.set(key, bytes)
this.settledBytes += bytes
}
this.trimSettled()
return
}
if (existing) {
return
}
if (this.liveBytes + command.bytes > this.maxMetadataBytes) {
throw new Error('Codex command metadata was not admitted before observation')
}
this.commands.set(key, command)
this.liveBytes += command.bytes
this.trimSettled()
}
tasks(
coveredThreads?: ReadonlySet<string>,
childLabel?: (threadId: string) => string | null
): AgentSessionBackgroundTask[] {
return [...this.commands.values()]
.filter((command) => !coveredThreads?.has(command.threadId))
.map(({ threadId, task }) => {
// The agent row carrying the child's name is gone by the time this row shows;
// unqualified it reads as a bare shell string with no owner. Resolved on read so
// a label registered after the command still lands.
const label = threadId === this.primaryThreadId ? null : childLabel?.(threadId)
return label
? { ...task, description: qualifiedDescription(label, task.description) }
: task
})
}
clear(): void {
this.commands.clear()
this.settled.clear()
this.liveBytes = 0
this.settledBytes = 0
}
private trimSettled(): void {
while (
this.settled.size > MAX_SETTLED_COMMANDS ||
this.retainedMetadataBytes > this.maxMetadataBytes
) {
const oldest = this.settled.entries().next().value
if (!oldest) {
break
}
this.settled.delete(oldest[0])
this.settledBytes -= oldest[1]
}
}
private parse(
event: CodexBackgroundTaskEvent
): { key: string; command: Command; completed: boolean } | null {
if (event.method !== 'item/started' && event.method !== 'item/completed') {
return null
}
const item = readCodexThreadItem(readRecord(event.params).item)
if (!item || !codexCommandOutlivesTurn(item)) {
return null
}
const key = JSON.stringify([event.threadId, item.id])
const completed = event.method === 'item/completed' || item.status !== 'inProgress'
const description = readString(item, 'command')
?.slice(0, MAX_DESCRIPTION_CHARS)
.replace(/\s+/g, ' ')
.trim()
const value = {
threadId: event.threadId,
task: {
id:
event.threadId === this.primaryThreadId
? `codex-command:primary:${encodeURIComponent(item.id)}`
: `codex-command:thread:${encodeURIComponent(event.threadId)}:${encodeURIComponent(item.id)}`,
kind: 'command' as const,
...(description ? { description } : {})
}
}
return {
key,
completed,
command: {
...value,
bytes:
Buffer.byteLength(key, 'utf8') + Buffer.byteLength(JSON.stringify(value), 'utf8') + 256
}
}
}
}
@@ -0,0 +1,72 @@
import type { NativeChatSubagentState } from '../../shared/native-chat-types'
import {
codexSubagentLabel,
isCodexRootAgentActivity,
readCodexSubagentActivity
} from './codex-subagent-activity'
import { codexChildTurnState } from './codex-subagent-executions'
import { readRecord } from './codex-item-field-readers'
import { readCodexThreadItem } from './codex-structured-item-translation'
import { readCodexTurnId } from './codex-structured-thread-facts'
export type CodexBackgroundTaskFrame =
| {
kind: 'subagent'
agentThreadId: string
label: string | null
parentTurnId: string | null | undefined
}
| {
kind: 'turn'
threadId: string
turnId: string
state: NativeChatSubagentState
}
export type CodexBackgroundTaskEvent = {
method: string
threadId: string
params: unknown
}
export function readCodexBackgroundTaskFrame(
event: CodexBackgroundTaskEvent,
primaryThreadId: string
): CodexBackgroundTaskFrame | null {
if (event.method === 'turn/started' || event.method === 'turn/completed') {
const turnId = readCodexTurnId(event.params)
if (turnId === null) {
return null
}
return {
kind: 'turn',
threadId: event.threadId,
turnId,
state:
event.method === 'turn/started'
? 'working'
: codexChildTurnState(readRecord(readRecord(event.params).turn).status)
}
}
if (event.method !== 'item/started' && event.method !== 'item/completed') {
return null
}
const item = readCodexThreadItem(readRecord(event.params).item)
const activity = item && readCodexSubagentActivity(item)
if (
!activity ||
activity.agentThreadId === primaryThreadId ||
isCodexRootAgentActivity(activity)
) {
return null
}
return {
kind: 'subagent',
agentThreadId: activity.agentThreadId,
label: codexSubagentLabel(activity),
parentTurnId:
activity.kind === 'started' || activity.kind === 'interacted'
? readCodexTurnId(event.params)
: undefined
}
}
@@ -0,0 +1,280 @@
import { describe, expect, it } from 'vitest'
import { CodexBackgroundTaskTracker } from './codex-background-task-tracker'
import {
readCodexBackgroundTaskFrame,
type CodexBackgroundTaskEvent
} from './codex-background-task-frames'
const PRIMARY = 'parent-thread'
const PARENT_TURN = 'parent-turn'
const CHILD = 'child-thread'
const CHILD_TURN = 'child-turn'
function turn(
method: 'turn/started' | 'turn/completed',
threadId: string,
turnId: string,
status = 'completed'
): CodexBackgroundTaskEvent {
return { method, threadId, params: { threadId, turn: { id: turnId, status } } }
}
function activity(
kind = 'started',
parentTurn = PARENT_TURN,
child = CHILD
): CodexBackgroundTaskEvent {
return {
method: 'item/started',
threadId: PRIMARY,
params: {
threadId: PRIMARY,
turnId: parentTurn,
item: {
type: 'subAgentActivity',
id: `activity-${kind}`,
kind,
agentThreadId: child,
agentPath: '/root/count_a'
}
}
}
}
function runningChild(): CodexBackgroundTaskTracker {
const tracker = new CodexBackgroundTaskTracker(PRIMARY)
tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN))
tracker.observe(turn('turn/started', CHILD, CHILD_TURN))
tracker.observe(activity())
return tracker
}
function command(threadId = PRIMARY, method = 'item/started'): CodexBackgroundTaskEvent {
return {
method,
threadId,
params: {
threadId,
turnId: PARENT_TURN,
item: {
type: 'commandExecution',
id: 'exec-1',
processId: '71831',
source: 'unifiedExecStartup',
command: 'sleep 90',
status: method === 'item/started' ? 'inProgress' : 'completed'
}
}
}
}
describe('readCodexBackgroundTaskFrame', () => {
it('reads activity as child metadata without inferring execution state', () => {
expect(readCodexBackgroundTaskFrame(activity('interacted'), PRIMARY)).toEqual({
kind: 'subagent',
agentThreadId: CHILD,
label: 'count_a',
parentTurnId: PARENT_TURN
})
})
it('reads a child turn with its own execution identity', () => {
expect(readCodexBackgroundTaskFrame(turn('turn/started', CHILD, CHILD_TURN), PRIMARY)).toEqual({
kind: 'turn',
threadId: CHILD,
turnId: CHILD_TURN,
state: 'working'
})
})
it('does not register the primary thread even when its activity path is missing', () => {
const event = activity('interacted', PARENT_TURN, PRIMARY)
;(event.params as { item: { agentPath?: string } }).item.agentPath = undefined
expect(readCodexBackgroundTaskFrame(event, PRIMARY)).toBeNull()
})
})
describe('CodexBackgroundTaskTracker child execution ownership', () => {
it('does not claim work from an activity item without a child turn', () => {
const tracker = new CodexBackgroundTaskTracker(PRIMARY)
tracker.observe(activity())
tracker.observe(activity('interacted'))
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.state).toBeNull()
})
it('reports an executing child only after the foreground turn ends', () => {
const tracker = runningChild()
expect(tracker.state).toBeNull()
expect(tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))).toBe(true)
expect(tracker.state).toEqual({
state: 'monitoring',
supportsStopAll: false,
tasks: [{ id: `codex-agent:${CHILD}`, kind: 'agent', description: 'count_a' }]
})
})
it('never settles a child when a primary turn ends', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
for (let index = 0; index < 300; index++) {
expect(tracker.observe(turn('turn/completed', PRIMARY, `later-${index}`))).toBe(false)
}
expect(tracker.state?.tasks).toHaveLength(1)
})
it.each(['completed', 'interrupted', 'failed'])(
'settles on the matching child turn %s',
(status) => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.observe(turn('turn/completed', CHILD, CHILD_TURN, status))).toBe(true)
expect(tracker.state).toBeNull()
}
)
it('does not mistake late activity completion for the current child execution', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
tracker.observe(activity('completed'))
expect(tracker.state?.tasks).toHaveLength(1)
})
it.each([PARENT_TURN, 'followup-parent'])(
'reports follow-up work in %s using the new child turn',
(parentTurn) => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
tracker.observe(turn('turn/started', PRIMARY, parentTurn))
tracker.observe(activity('interacted', parentTurn))
expect(tracker.state).toBeNull()
tracker.observe(turn('turn/started', CHILD, 'followup-child-turn'))
tracker.observe(turn('turn/completed', PRIMARY, parentTurn))
expect(tracker.state?.tasks).toHaveLength(1)
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))
tracker.observe(turn('turn/started', CHILD, CHILD_TURN))
tracker.observe(activity('completed'))
expect(tracker.state?.tasks).toHaveLength(1)
tracker.observe(turn('turn/completed', CHILD, 'followup-child-turn'))
expect(tracker.state).toBeNull()
}
)
it('keeps idle send_message activity out of the strip', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
tracker.observe(activity('interacted', 'message-parent'))
tracker.observe(turn('turn/completed', PRIMARY, 'message-parent'))
expect(tracker.state).toBeNull()
})
it('does not invent another execution for a message to a working child', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
tracker.observe(activity('interacted', 'message-parent'))
tracker.observe(turn('turn/completed', PRIMARY, 'message-parent'))
expect(tracker.state?.tasks).toHaveLength(1)
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))
expect(tracker.state).toBeNull()
})
it('retains completion delivered before child registration', () => {
const tracker = new CodexBackgroundTaskTracker(PRIMARY)
tracker.observe(turn('turn/started', CHILD, CHILD_TURN))
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))
tracker.observe(activity())
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.state).toBeNull()
})
it('publishes no extra state for duplicate owner or metadata events', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.observe(turn('turn/started', CHILD, CHILD_TURN))).toBe(false)
expect(tracker.observe({ ...activity(), method: 'item/completed' })).toBe(false)
expect(tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))).toBe(true)
expect(tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))).toBe(false)
})
it('bounds retained child history while allowing repeated completed runs', () => {
const tracker = new CodexBackgroundTaskTracker(PRIMARY)
tracker.observe(activity())
for (let index = 0; index < 300; index++) {
const id = `child-turn-${index}`
tracker.observe(turn('turn/started', CHILD, id))
expect(tracker.state?.tasks).toHaveLength(1)
tracker.observe(turn('turn/completed', CHILD, id))
expect(tracker.state).toBeNull()
}
})
it('clears the roster at session teardown', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.clear()).toBe(true)
expect(tracker.state).toBeNull()
expect(tracker.clear()).toBe(false)
})
})
describe('CodexBackgroundTaskTracker command integration', () => {
it('keeps a primary shell visible after the turn until its own completion', () => {
const tracker = new CodexBackgroundTaskTracker(PRIMARY)
tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN))
tracker.observe(command())
expect(tracker.state).toBeNull()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.state?.tasks).toEqual([
{ id: 'codex-command:primary:exec-1', kind: 'command', description: 'sleep 90' }
])
tracker.observe(command(PRIMARY, 'item/completed'))
expect(tracker.state).toBeNull()
})
it('reveals a child shell only after the child execution finishes', () => {
const tracker = runningChild()
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
tracker.observe(command(CHILD))
expect(tracker.state?.tasks).toHaveLength(1)
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN, 'interrupted'))
expect(tracker.state?.tasks).toEqual([
{
id: `codex-command:thread:${CHILD}:exec-1`,
kind: 'command',
description: 'count_a — sleep 90'
}
])
tracker.observe(command(CHILD, 'item/completed'))
expect(tracker.state).toBeNull()
})
it('leaves a primary shell unqualified', () => {
const tracker = runningChild()
tracker.observe(command(PRIMARY))
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
expect(tracker.state?.tasks).toContainEqual({
id: 'codex-command:primary:exec-1',
kind: 'command',
description: 'sleep 90'
})
})
it('names a child shell whose label only arrives after the command', () => {
const tracker = new CodexBackgroundTaskTracker(PRIMARY)
tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN))
tracker.observe(turn('turn/started', CHILD, CHILD_TURN))
tracker.observe(command(CHILD))
tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))
tracker.observe(activity())
tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))
expect(tracker.state?.tasks).toEqual([
{
id: `codex-command:thread:${CHILD}:exec-1`,
kind: 'command',
description: 'count_a — sleep 90'
}
])
})
})
@@ -0,0 +1,96 @@
import type {
AgentSessionBackgroundTask,
AgentSessionBackgroundTaskState
} from '../../shared/agent-session-wire'
import {
readCodexBackgroundTaskFrame,
type CodexBackgroundTaskEvent
} from './codex-background-task-frames'
import { CodexSubagentExecutions } from './codex-subagent-executions'
import { CodexBackgroundCommandTracker } from './codex-background-command-tracker'
import { boundSubagentField } from './codex-subagent-group-body'
/** Projects the same child execution facts the durable roster consumes. */
export class CodexBackgroundTaskTracker {
private primaryTurnId: string | null = null
private publishedFingerprint = '[]'
private publishedState: AgentSessionBackgroundTaskState | null = null
private readonly commands: CodexBackgroundCommandTracker
constructor(
private readonly primaryThreadId: string,
private readonly executions = new CodexSubagentExecutions()
) {
this.commands = new CodexBackgroundCommandTracker(primaryThreadId)
}
get state(): AgentSessionBackgroundTaskState | null {
// Journal admission precedes observe; readers must not see its pending facts.
return this.publishedState
}
canObserve(event: CodexBackgroundTaskEvent): boolean {
return this.commands.canObserve(event)
}
observe(event: CodexBackgroundTaskEvent): boolean {
const itemEvent = event.method === 'item/started' || event.method === 'item/completed'
if (itemEvent) {
this.commands.observe(event)
}
const frame = readCodexBackgroundTaskFrame(event, this.primaryThreadId)
if (!frame) {
return itemEvent ? this.refresh() : false
}
if (frame.kind === 'subagent') {
this.executions.register(frame.agentThreadId, frame.label, frame.parentTurnId)
} else if (frame.threadId === this.primaryThreadId) {
if (frame.state === 'working') {
this.primaryTurnId = frame.turnId
} else if (frame.turnId === this.primaryTurnId) {
this.primaryTurnId = null
}
} else {
this.executions.observeTurn(frame.threadId, frame.turnId, frame.state)
}
return this.refresh()
}
clear(): boolean {
this.executions.clear()
this.commands.clear()
this.primaryTurnId = null
return this.refresh()
}
private tasks(): AgentSessionBackgroundTask[] {
if (this.primaryTurnId !== null) {
return []
}
const children = this.executions.workingChildren()
const agents: AgentSessionBackgroundTask[] = children.map((child, index) => ({
id: `codex-agent:${child.agentThreadId}`,
kind: 'agent',
...(child.label ? { description: boundSubagentField(child.label, index) } : {})
}))
return [
...agents,
...this.commands.tasks(new Set(children.map((child) => child.agentThreadId)), (threadId) =>
this.executions.label(threadId)
)
]
}
private refresh(): boolean {
const tasks = this.tasks()
const fingerprint = JSON.stringify(tasks)
if (fingerprint === this.publishedFingerprint) {
return false
}
this.publishedFingerprint = fingerprint
this.publishedState = tasks.length
? { state: 'monitoring', tasks, supportsStopAll: false }
: null
return true
}
}
@@ -0,0 +1,6 @@
import type { CodexThreadItem } from './codex-structured-item-translation'
/** Persistent exec has its own process-exit notification, independent of a turn. */
export function codexCommandOutlivesTurn(item: CodexThreadItem): boolean {
return item.type === 'commandExecution' && item.source === 'unifiedExecStartup'
}
@@ -0,0 +1,107 @@
import { codexCommandOutlivesTurn } from './codex-command-lifecycle'
import {
MAX_CODEX_ITEM_STREAM_ITEM_BYTES,
MAX_CODEX_ITEM_STREAM_STATES
} from './codex-structured-item-stream-bounds'
import type { CodexItemStreamState } from './codex-structured-item-stream-contracts'
// Preserve the previous metadata ceiling while letting small live commands share it.
export const MAX_CODEX_ITEM_STREAM_METADATA_BYTES =
MAX_CODEX_ITEM_STREAM_STATES * MAX_CODEX_ITEM_STREAM_ITEM_BYTES
type RetainedState = { state: CodexItemStreamState; bytes: number; persistent: boolean }
export class CodexItemStreamRetention {
private readonly states = new Map<string, RetainedState>()
private bytes = 0
private persistentBytes = 0
private persistentCount = 0
constructor(private readonly maxBytes = MAX_CODEX_ITEM_STREAM_METADATA_BYTES) {}
get retainedBytes(): number {
return this.bytes
}
get size(): number {
return this.states.size
}
get persistentSize(): number {
return this.persistentCount
}
get overCapacity(): boolean {
return (
this.bytes > this.maxBytes ||
this.states.size - this.persistentCount > MAX_CODEX_ITEM_STREAM_STATES
)
}
get(key: string): CodexItemStreamState | undefined {
return this.states.get(key)?.state
}
isPersistent(key: string): boolean {
return this.states.get(key)?.persistent === true
}
canRetain(key: string, state: CodexItemStreamState): boolean {
const previous = this.states.get(key)
return (
this.persistentBytes -
(previous?.persistent ? previous.bytes : 0) +
this.stateBytes(key, state) <=
this.maxBytes
)
}
retain(key: string, state: CodexItemStreamState): boolean {
if (!this.canRetain(key, state)) {
return false
}
this.forget(key)
const bytes = this.stateBytes(key, state)
const persistent = codexCommandOutlivesTurn(state.item)
this.states.set(key, { state, bytes, persistent })
this.bytes += bytes
if (persistent) {
this.persistentBytes += bytes
this.persistentCount += 1
}
return true
}
oldestEvictable(): string | undefined {
for (const [key, entry] of this.states) {
if (!entry.persistent) {
return key
}
}
return undefined
}
forget(key: string): void {
const entry = this.states.get(key)
if (!entry) {
return
}
this.bytes -= entry.bytes
if (entry.persistent) {
this.persistentBytes -= entry.bytes
this.persistentCount -= 1
}
this.states.delete(key)
}
clear(): void {
this.states.clear()
this.bytes = 0
this.persistentBytes = 0
this.persistentCount = 0
}
private stateBytes(key: string, state: CodexItemStreamState): number {
return Buffer.byteLength(key, 'utf8') + Buffer.byteLength(JSON.stringify(state), 'utf8') + 256
}
}
@@ -0,0 +1,219 @@
import { describe, expect, it, vi } from 'vitest'
import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types'
import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key'
import { CodexBackgroundCommandTracker } from './codex-background-command-tracker'
import {
CodexItemStreamRetention,
MAX_CODEX_ITEM_STREAM_METADATA_BYTES
} from './codex-item-stream-retention'
import { CodexJournalItems } from './codex-structured-journal-items'
import { settleCodexJournalTurn } from './codex-structured-journal-settlement'
function command(threadId: string, id: string, method = 'item/started') {
return {
threadId,
method,
params: {
turnId: 'turn',
item: {
type: 'commandExecution',
id,
source: 'unifiedExecStartup',
command: `sleep 30 # ${threadId}/${id}`,
cwd: '/workspace',
status: method === 'item/completed' ? 'completed' : 'inProgress',
...(method === 'item/completed' ? { exitCode: 0, aggregatedOutput: 'BEFORE\nAFTER\n' } : {})
}
}
}
}
function fixture(maxMetadataBytes?: number) {
const rows = new Map<string, AgentJournalItemBody>()
const scheduled = new Set<() => void>()
const sink = {
appendItem: (
identity: Parameters<typeof agentJournalItemKey>[0],
body: AgentJournalItemBody
) => {
rows.set(agentJournalItemKey(identity), body)
},
appendTombstone() {},
publish() {}
}
const items = new CodexJournalItems(
{
sink,
maxMetadataBytes,
schedule: (run) => {
scheduled.add(run)
return () => {
scheduled.delete(run)
}
}
},
() => 'turn',
() => {}
)
return { items, sink, rows, scheduled }
}
describe('persistent command retention', () => {
it('does not rebuild unchanged persistent output on every later lifecycle flush', () => {
const { items } = fixture()
const event = command('root', 'quiet')
items.handle(event)
items.streams.handle('root', 'item/commandExecution/outputDelta', {
itemId: 'quiet',
delta: 'retained-prefix'
})
items.streams.flush()
const originalJoin = Array.prototype.join
let retainedJoins = 0
const spy = vi
.spyOn(Array.prototype, 'join')
.mockImplementation(function (this: unknown[], separator) {
if (this[0] === 'retained-prefix') {
retainedJoins += 1
}
return originalJoin.call(this, separator)
})
try {
for (let index = 0; index < 100; index += 1) {
items.streams.flush()
}
} finally {
spy.mockRestore()
items.dispose()
}
expect(retainedJoins).toBe(0)
})
it('retains 448 live commands through completed turns, late output, and process completion', () => {
const { items, sink, rows, scheduled } = fixture()
const tracker = new CodexBackgroundCommandTracker('thread-0')
const events = Array.from({ length: 7 }, (_, thread) =>
Array.from({ length: 64 }, (_, index) => command(`thread-${thread}`, `exec-${index}`))
).flat()
for (const event of events) {
expect(tracker.canObserve(event)).toBe(true)
expect(items.handle(event)).toMatchObject({ admission: { accepted: true } })
tracker.observe(event)
expect(
items.streams.handle(event.threadId, 'item/commandExecution/outputDelta', {
turnId: 'turn',
itemId: event.params.item.id,
delta: 'BEFORE\n'
}).admission
).toEqual({ accepted: true })
}
for (let thread = 0; thread < 7; thread += 1) {
expect(
settleCodexJournalTurn({
sessionId: 'session',
threadId: `thread-${thread}`,
turnId: 'turn',
sink,
streams: items.streams,
activeItems: items.activeItems
})
).toEqual({ accepted: true })
}
expect(items.activeItems.size).toBe(448)
expect(items.streams.persistentCount).toBe(448)
expect(tracker.tasks()).toHaveLength(448)
expect(tracker.retainedMetadataBytes).toBeLessThan(256 * 1024)
for (const event of events) {
items.streams.handle(event.threadId, 'item/commandExecution/outputDelta', {
turnId: 'turn',
itemId: event.params.item.id,
delta: 'AFTER\n'
})
}
expect(items.streams.flush()).toBe(true)
for (const event of events) {
const key = agentJournalItemKey({
provider: 'orca',
clientMessageId: `codex-item:${event.threadId}:${event.params.item.id}`
})
expect(rows.get(key)).toMatchObject({
state: 'running',
input: { command: event.params.item.command, cwd: '/workspace' },
output: { head: 'BEFORE\nAFTER\n' }
})
const completed = command(event.threadId, event.params.item.id, 'item/completed')
expect(items.handle(completed)).toMatchObject({ admission: { accepted: true } })
tracker.observe(completed)
expect(rows.get(key)).toMatchObject({
state: 'completed',
output: { head: 'BEFORE\nAFTER\n' }
})
expect(items.streams.snapshot(event.threadId, event.params.item.id)).toBeNull()
}
expect(items.activeItems.size).toBe(0)
expect(items.streams.persistentCount).toBe(0)
expect(tracker.tasks()).toEqual([])
expect(tracker.retainedMetadataBytes).toBeLessThan(64 * 1024)
items.dispose()
tracker.clear()
expect(tracker.retainedMetadataBytes).toBe(0)
expect(scheduled.size).toBe(0)
})
it('rejects command metadata exhaustion before appending or evicting live state and frees it on completion', () => {
const { items, rows } = fixture(800)
const first = command('root', 'first')
const second = command('root', 'second')
expect(items.handle(first)).toMatchObject({ admission: { accepted: true } })
const prior = [...rows]
expect(items.handle(second)).toMatchObject({ admission: { accepted: false, reason: 'failed' } })
expect([...rows]).toEqual(prior)
expect(items.activeItems.size).toBe(1)
expect(items.handle(command('root', 'first', 'item/completed'))).toMatchObject({
admission: { accepted: true }
})
expect(items.handle(second)).toMatchObject({ admission: { accepted: true } })
items.dispose()
})
it('accounts metadata bytes instead of interpreting the item count as liveness', () => {
const retention = new CodexItemStreamRetention()
for (let index = 0; index < 448; index += 1) {
const item = command('root', `exec-${index}`).params.item
expect(
retention.retain(item.id, {
item,
identity: { provider: 'orca', clientMessageId: item.id }
})
).toBe(true)
}
expect(retention.size).toBe(448)
expect(retention.retainedBytes).toBeLessThan(256 * 1024)
expect(retention.retainedBytes).toBeLessThan(MAX_CODEX_ITEM_STREAM_METADATA_BYTES)
expect(retention.overCapacity).toBe(false)
expect(retention.oldestEvictable()).toBeUndefined()
retention.clear()
expect(retention.retainedBytes).toBe(0)
expect(retention.persistentSize).toBe(0)
})
it('retains startup provenance when large command metadata is bounded', () => {
const { items, sink } = fixture()
const event = command('root', 'large')
event.params.item.command = 'x'.repeat(128 * 1024)
expect(items.handle(event)).toMatchObject({ admission: { accepted: true } })
expect(
settleCodexJournalTurn({
sessionId: 'session',
threadId: 'root',
turnId: 'turn',
sink,
streams: items.streams,
activeItems: items.activeItems
})
).toEqual({ accepted: true })
expect(items.activeItems.size).toBe(1)
expect(items.streams.persistentCount).toBe(1)
items.dispose()
})
})
@@ -1,4 +1,4 @@
import { describe, expect, it } from 'vitest'
import { describe, expect, it, vi } from 'vitest'
import {
compareCodexSessionBackfillDates,
expandCodexSessionBackfillDatesThroughToday,
@@ -9,6 +9,7 @@ import {
parseCodexSessionBackfillDates,
subtractCodexSessionBackfillDates
} from './codex-session-backfill-scan-dates'
import type { CodexSessionBackfillDate } from './codex-session-backfill-types'
describe('codex session backfill scan dates', () => {
it('reads UTC parts so a local evening never lands on the wrong directory', () => {
@@ -100,3 +101,63 @@ describe('codex session backfill scan dates', () => {
).toBeNull()
})
})
describe('bounded backfill range construction', () => {
it('does not allocate rejected dates for a decades-old pending marker', () => {
const advance = vi.spyOn(Date.prototype, 'setUTCDate')
try {
expect(
expandCodexSessionBackfillDatesThroughToday(
[['2000', '01', '01']],
['2026', '09', '07'],
31
)
).toBeNull()
expect(advance).not.toHaveBeenCalled()
} finally {
advance.mockRestore()
}
})
it('keeps exact, fractional, leap-day and future-clock bounds', () => {
const dates = [['2024', '02', '28']] as [string, string, string][]
expect(expandCodexSessionBackfillDatesThroughToday(dates, ['2024', '03', '01'], 3)).toEqual([
['2024', '02', '28'],
['2024', '02', '29'],
['2024', '03', '01']
])
expect(expandCodexSessionBackfillDatesThroughToday(dates, ['2024', '03', '01'], 2.5)).toBeNull()
expect(
expandCodexSessionBackfillDatesThroughToday([['2024', '03', '01']], ['2024', '02', '28'], 3)
).toEqual(expandCodexSessionBackfillDatesThroughToday(dates, ['2024', '03', '01'], 3))
})
// The arithmetic cardinality gate must admit and reject exactly what enumerating the range
// would, on every calendar edge that has ever broken a day count: leap days, century rules,
// year rollover, and the DST switches the UTC-only arithmetic has to stay indifferent to.
it.each<[string, CodexSessionBackfillDate, CodexSessionBackfillDate]>([
['leap February', ['2024', '02', '27'], ['2024', '03', '02']],
['non-leap February', ['2023', '02', '27'], ['2023', '03', '02']],
['US spring-forward', ['2024', '03', '09'], ['2024', '03', '11']],
['US fall-back', ['2024', '11', '02'], ['2024', '11', '04']],
['EU spring-forward', ['2025', '03', '29'], ['2025', '03', '31']],
['southern-hemisphere DST', ['2025', '04', '05'], ['2025', '04', '07']],
['year rollover', ['2024', '12', '30'], ['2025', '01', '02']],
['leap century', ['1999', '12', '31'], ['2000', '01', '02']],
['non-leap century', ['2100', '02', '27'], ['2100', '03', '02']],
['30-day month end', ['2026', '04', '29'], ['2026', '05', '02']],
['single day', ['2026', '09', '07'], ['2026', '09', '07']]
])('matches the enumerated range at the %s cap boundary', (_label, from, to) => {
const start = new Date(Date.UTC(Number(from[0]), Number(from[1]) - 1, Number(from[2])))
const end = new Date(Date.UTC(Number(to[0]), Number(to[1]) - 1, Number(to[2])))
const enumerated = getCodexSessionBackfillDatesBetween(start, end)
const pending = [from]
expect(expandCodexSessionBackfillDatesThroughToday(pending, to, enumerated.length)).toEqual(
enumerated
)
expect(
expandCodexSessionBackfillDatesThroughToday(pending, to, enumerated.length - 1)
).toBeNull()
})
})
@@ -99,8 +99,13 @@ export function expandCodexSessionBackfillDatesThroughToday(
return []
}
const bounds = mergeCodexSessionBackfillDates(dates, [today])
const range = getCodexSessionBackfillDatesBetween(toUtcDate(bounds[0]), toUtcDate(bounds.at(-1)!))
return range.length > maxDates ? null : range
const first = toUtcDate(bounds[0])
const last = toUtcDate(bounds.at(-1)!)
const dateCount = (last.getTime() - first.getTime()) / 86_400_000 + 1
if (dateCount > maxDates) {
return null
}
return getCodexSessionBackfillDatesBetween(first, last)
}
function toUtcDate([year, month, day]: readonly string[]): Date {
@@ -30,6 +30,7 @@ export function boundStreamItem(item: Record<string, unknown>): Record<string, u
return {
type: item.type,
id: item.id,
...(typeof item.source === 'string' ? { source: item.source } : {}),
...(typeof item.command === 'string' ? { command: item.command.slice(0, 4096) } : {}),
...(typeof item.cwd === 'string' ? { cwd: item.cwd.slice(0, 4096) } : {}),
...(typeof item.status === 'string' ? { status: item.status } : {}),
@@ -13,6 +13,7 @@ export type CodexItemStreamDeps = {
coalesceMs?: number
maxRetainedBytes?: number
maxTotalRetainedBytes?: number
maxMetadataBytes?: number
schedule?: AgentSessionDeltaCoalescerDeps['schedule']
}
@@ -36,7 +37,9 @@ export type CodexStructuredItemStreamHandleResult = {
}
export type CodexStructuredItemStreams = {
track: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => void
readonly persistentCount: number
canTrack: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => boolean
track: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => boolean
handle: (
threadId: string,
method: string,
+40 -28
View File
@@ -1,5 +1,6 @@
import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key'
import { createAgentSessionDeltaCoalescer } from '../native-chat/agent-session-wire/agent-session-delta-coalescer'
import { CodexItemStreamRetention } from './codex-item-stream-retention'
import {
codexJournalItem,
codexStreamingJournalItem,
@@ -10,7 +11,6 @@ import {
MAX_CODEX_ITEM_STREAM_PENDING_PATCHES,
MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES,
MAX_CODEX_ITEM_STREAM_RETAINED_BYTES,
MAX_CODEX_ITEM_STREAM_STATES,
boundStreamItem,
pendingPatchBytes
} from './codex-structured-item-stream-bounds'
@@ -47,8 +47,9 @@ export {
export function createCodexStructuredItemStreams(
deps: CodexItemStreamDeps
): CodexStructuredItemStreams {
const states = new Map<string, CodexItemStreamState>()
const states = new CodexItemStreamRetention(deps.maxMetadataBytes)
const checkpointLengths = new Map<string, number>()
const pendingCheckpoints = new Set<string>()
// Patch updates are authoritative item snapshots. Keep the latest rejected
// snapshot until the journal admits it; unlike streamed deltas, there is no
// coalescer timer to retry these events for us.
@@ -57,8 +58,9 @@ export function createCodexStructuredItemStreams(
const forgetState = (key: string): void => {
coalescer.forget(key)
states.delete(key)
states.forget(key)
checkpointLengths.delete(key)
pendingCheckpoints.delete(key)
const pending = pendingPatches.get(key)
if (pending) {
retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending))
@@ -67,8 +69,8 @@ export function createCodexStructuredItemStreams(
}
const trimStates = (): void => {
while (states.size > MAX_CODEX_ITEM_STREAM_STATES) {
const oldest = states.keys().next().value
while (states.overCapacity) {
const oldest = states.oldestEvictable()
if (typeof oldest !== 'string') {
break
}
@@ -124,6 +126,7 @@ export function createCodexStructuredItemStreams(
const state = states.get(key)
if (state && append(state, text)) {
checkpointLengths.set(key, text.length)
pendingCheckpoints.delete(key)
return true
}
return false
@@ -133,6 +136,7 @@ export function createCodexStructuredItemStreams(
windowMs: deps.coalesceMs,
maxRetainedBytes: deps.maxRetainedBytes,
maxTotalRetainedBytes: deps.maxTotalRetainedBytes,
isProtected: (key) => states.isPersistent(key),
schedule: deps.schedule,
emit: (key, text) => {
return persist(key, text, false)
@@ -144,7 +148,7 @@ export function createCodexStructuredItemStreams(
itemId: string,
type: string,
params: unknown
): CodexItemStreamState => {
): CodexItemStreamState | null => {
const key = codexStructuredItemKey(threadId, itemId)
const existing = states.get(key)
if (existing) {
@@ -152,36 +156,25 @@ export function createCodexStructuredItemStreams(
}
const item = { type, id: itemId }
const state = { item, identity: deps.identityFor(threadId, params, item) }
states.set(key, state)
if (!states.retain(key, state)) {
return null
}
trimStates()
return state
}
const flush = (): boolean => {
let flushed = coalescer.flushAll()
for (const key of states.keys()) {
for (const key of pendingCheckpoints) {
const snapshot = coalescer.snapshot(key)
if (snapshot && checkpointLengths.get(key) !== snapshot.text.length) {
flushed = persist(key, snapshot.text, true) && flushed
} else {
pendingCheckpoints.delete(key)
}
}
for (const [key, pending] of pendingPatches) {
const admission = deps.sink.tryAppendItem
? deps.sink.tryAppendItem(pending.identity, pending.body)
: (deps.sink.appendItem(pending.identity, pending.body), { accepted: true as const })
if (!admission.accepted) {
flushed = false
continue
}
const published = deps.sink.tryPublish
? deps.sink.tryPublish()
: (deps.sink.publish(), { accepted: true as const })
if (!published.accepted) {
flushed = false
continue
}
retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending))
pendingPatches.delete(key)
for (const key of pendingPatches.keys()) {
flushed = flushPatch(key).accepted && flushed
}
return flushed
}
@@ -209,11 +202,21 @@ export function createCodexStructuredItemStreams(
}
return {
get persistentCount() {
return states.persistentSize
},
canTrack: (threadId, item, identity) =>
states.canRetain(codexStructuredItemKey(threadId, item.id), {
item: boundStreamItem(item) as CodexThreadItem,
identity
}),
track: (threadId, item, identity) => {
const key = codexStructuredItemKey(threadId, item.id)
states.delete(key)
states.set(key, { item: boundStreamItem(item) as CodexThreadItem, identity })
if (!states.retain(key, { item: boundStreamItem(item) as CodexThreadItem, identity })) {
return false
}
trimStates()
return true
},
handle: (threadId, method, params) => {
const paramsRecord = readCodexItemStreamRecord(params)
@@ -230,6 +233,9 @@ export function createCodexStructuredItemStreams(
const key = codexStructuredItemKey(threadId, itemId)
const streamFlushed = coalescer.flush(key)
const state = ensureState(threadId, itemId, 'fileChange', params)
if (!state) {
return { handled: true, admission: { accepted: false, reason: 'failed' } }
}
state.item = { ...state.item, changes: paramsRecord.changes }
const translated = codexJournalItem(state.item)
if (translated.body) {
@@ -269,9 +275,14 @@ export function createCodexStructuredItemStreams(
return { handled: true, admission: { accepted: true } }
}
const state = ensureState(threadId, itemId, type ?? 'reasoning', params)
if (!state) {
return { handled: true, admission: { accepted: false, reason: 'failed' } }
}
const delta = method === REASONING_PART_METHOD ? '\n' : paramsRecord.delta
if (typeof delta === 'string') {
const accepted = coalescer.append(codexStructuredItemKey(threadId, state.item.id), delta)
const key = codexStructuredItemKey(threadId, state.item.id)
pendingCheckpoints.add(key)
const accepted = coalescer.append(key, delta)
if (!accepted) {
return { handled: true, admission: { accepted: false, reason: 'backpressure' } }
}
@@ -287,6 +298,7 @@ export function createCodexStructuredItemStreams(
coalescer.dispose()
states.clear()
checkpointLengths.clear()
pendingCheckpoints.clear()
pendingPatches.clear()
retainedPatchBytes = 0
},
@@ -584,6 +584,16 @@ describe('codex item bodies', () => {
})
})
it('preserves plan prose documents byte-for-byte as status text', () => {
const text =
' # Implementation plan\r\n\r\n- [ ] Preserve prose\r\n- [x] Keep café → 日本語\r\n\r\n```ts\r\nconst task = "pending"\r\n```\r\n '
expect(codexJournalItem({ type: 'plan', id: 'plan-document', text })).toEqual({
body: { kind: 'status', text, presentation: 'plan-document' },
handled: true
})
})
it('renders reasoning as status and exposes an unknown item as a provider frame', () => {
expect(codexItemBody({ type: 'reasoning', id: 'r', text: 'thinking' })).toEqual({
kind: 'status',
@@ -1,11 +1,13 @@
import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer'
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter'
import type { CodexSubagentExecutions } from './codex-subagent-executions'
export type CodexJournalTranslatorDeps = {
sink: StructuredAgentSessionEventSink
bindPromptItemId?: (journalItemId: string, threadId: string, promptKey: string) => void
primaryThreadId?: () => string | null
subagentExecutions?: CodexSubagentExecutions
coalesceMs?: number
maxRetainedBytes?: number
schedule?: AgentSessionDeltaCoalescerDeps['schedule']
@@ -11,7 +11,8 @@ import {
type CodexThreadItem
} from './codex-structured-item-translation'
import { createCodexStructuredItemStreams } from './codex-structured-item-streams'
import { codexStructuredItemKey } from './codex-structured-item-stream-bounds'
import { boundStreamItem, codexStructuredItemKey } from './codex-structured-item-stream-bounds'
import { codexCommandOutlivesTurn } from './codex-command-lifecycle'
import type {
CodexItemTranslation,
CodexJournalTranslationAdmission,
@@ -40,7 +41,7 @@ export class CodexJournalItems {
private readonly deps: Pick<
CodexJournalTranslatorDeps,
'sink' | 'coalesceMs' | 'maxRetainedBytes' | 'schedule'
>,
> & { maxMetadataBytes?: number },
private readonly activeTurn: (threadId: string) => string | null,
private readonly suppress: (threadId: string, turnId: string) => void
) {
@@ -49,6 +50,7 @@ export class CodexJournalItems {
coalesceMs: deps.coalesceMs,
maxRetainedBytes: deps.maxRetainedBytes,
schedule: deps.schedule,
maxMetadataBytes: deps.maxMetadataBytes,
identityFor: (threadId, params, item) => {
const turnId = readCodexTurnId(params) ?? this.activeTurn(threadId)
return this.identityFor(threadId, turnId, item)
@@ -81,6 +83,12 @@ export class CodexJournalItems {
if (item.type === 'contextCompaction' && event.method === 'item/started') {
return { handled: true, admission: CODEX_JOURNAL_ADMITTED }
}
if (
event.method !== 'item/completed' &&
!this.streams.canTrack(event.threadId, item, identity)
) {
return { handled: true, admission: { accepted: false, reason: 'failed' } }
}
const translated = codexJournalItem(item)
const command = readCodexJournalString(item, 'command')
if (command) {
@@ -157,12 +165,15 @@ export class CodexJournalItems {
item: CodexThreadItem,
identity: AgentJournalItemIdentity
): void {
this.streams.track(threadId, item, identity)
const retainedItem = codexCommandOutlivesTurn(item)
? (boundStreamItem(item) as CodexThreadItem)
: item
this.streams.track(threadId, retainedItem, identity)
this.activeItems.set(codexStructuredItemKey(threadId, item.id), {
threadId,
turnId,
identity,
item
item: retainedItem
})
}
@@ -194,8 +205,10 @@ export class CodexJournalItems {
}
private trimActiveState(): CodexJournalTranslationAdmission {
while (this.activeItems.size > MAX_CODEX_ACTIVE_ITEMS) {
const oldest = this.activeItems.keys().next().value
while (this.activeItems.size - this.streams.persistentCount > MAX_CODEX_ACTIVE_ITEMS) {
const oldest = [...this.activeItems].find(
([, active]) => !codexCommandOutlivesTurn(active.item)
)?.[0]
if (typeof oldest !== 'string') {
break
}
@@ -20,6 +20,7 @@ import {
} from './codex-structured-item-translation'
import type { CodexStructuredItemStreams } from './codex-structured-item-streams'
import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter'
import { codexCommandOutlivesTurn } from './codex-command-lifecycle'
export type CodexActiveJournalItem = {
threadId: string
@@ -118,6 +119,9 @@ export function settleCodexJournalTurn(input: {
if (active.threadId !== input.threadId || active.turnId !== input.turnId) {
continue
}
if (codexCommandOutlivesTurn(active.item)) {
continue
}
const streamed = input.streams.snapshot(active.threadId, active.item.id)
const translated = streamed
? codexStreamingJournalItem(active.item, streamed.text)
@@ -6,6 +6,8 @@
* than as the shape checks each arm performs.
*/
import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter'
import type { CodexJournalItems } from './codex-structured-journal-items'
import type { CodexJournalTranslationAdmission } from './codex-structured-journal-contracts'
import { settleCodexOversizedNotification } from './codex-structured-journal-settlement'
import {
@@ -41,3 +43,23 @@ export function settleCodexOversizedNotificationFrame(input: {
})
: null
}
export function createCodexOversizedNotificationSettler(
deps: { sink: OversizedInput['sink'] },
items: Pick<CodexJournalItems, 'streams' | 'activeItems'>
) {
return settleOversizedNotification
/** Settles the item a notification the transport refused to carry left
* mid-flight; null when the frame is not one. */
function settleOversizedNotification(
event: Extract<CodexStructuredSessionEvent, { type: 'provider-frame' }>
): CodexJournalTranslationAdmission | null {
return settleCodexOversizedNotificationFrame({
...event,
sink: deps.sink,
streams: items.streams,
activeItems: items.activeItems
})
}
}
@@ -59,6 +59,22 @@ function deliverActivity(
translator: ReturnType<typeof createCodexJournalTranslator>,
params: unknown
): void {
const item = (params as { item: { kind: string; agentThreadId: string } }).item
if (item.kind === 'started' || item.kind === 'completed') {
translator.handle({
type: 'notification',
sessionId: SESSION_ID,
threadId: item.agentThreadId,
method: item.kind === 'started' ? 'turn/started' : 'turn/completed',
params: {
threadId: item.agentThreadId,
turn: {
id: `execution:${item.agentThreadId}`,
status: item.kind === 'started' ? 'inProgress' : 'completed'
}
}
})
}
translator.handle(notification('item/started', params))
translator.handle(notification('item/completed', params))
}
@@ -19,7 +19,7 @@ import {
settleCodexJournalSession,
settleCodexJournalTurn
} from './codex-structured-journal-settlement'
import { settleCodexOversizedNotificationFrame } from './codex-structured-journal-translation-frames'
import { createCodexOversizedNotificationSettler } from './codex-structured-journal-translation-frames'
import { restoreCodexJournalThread } from './codex-structured-journal-translation-restore'
import { CodexJournalActiveTurns } from './codex-structured-journal-translation-turn-state'
import { publishCodexTurnLifecycle } from './codex-structured-journal-translation-turns'
@@ -58,13 +58,15 @@ export function createCodexJournalTranslator(
(threadId) => activeTurns.current(threadId),
(threadId, turnId) => genericFrames.suppress(threadId, turnId)
)
const settleOversizedNotification = createCodexOversizedNotificationSettler(deps, items)
const prompts = new CodexJournalPrompts(deps, (threadId, itemId) =>
items.detailFor(threadId, itemId)
)
const subagents = new CodexSubagentRoster({
sink: deps.sink,
primaryThreadId: () => deps.primaryThreadId?.() ?? null,
activeTurn: (threadId) => activeTurns.current(threadId)
activeTurn: (threadId) => activeTurns.current(threadId),
...(deps.subagentExecutions ? { executions: deps.subagentExecutions } : {})
})
const flushStreams = (): CodexJournalTranslationAdmission =>
items.streams.flush() ? CODEX_JOURNAL_ADMITTED : { accepted: false, reason: 'backpressure' }
@@ -174,16 +176,17 @@ export function createCodexJournalTranslator(
}
return genericFrames.appendUnhandled(event.kind, event.payload, event.threadId)
}
if (event.method === 'turn/started') {
return startTurn(event)
if (event.method === 'turn/started' || event.method === 'turn/completed') {
const childAdmission = subagents.handleTurnEvent(event)
if (!childAdmission.accepted) {
return childAdmission
}
return event.method === 'turn/started' ? startTurn(event) : completeTurn(event)
}
const compaction = compactions.handle(event)
if (compaction) {
return publishActivity(event, compaction)
}
if (event.method === 'turn/completed') {
return completeTurn(event)
}
if (event.method === CODEX_TOKEN_USAGE_METHOD) {
// Classified `status-chrome`, so the generic-frame path swallows it
// before the journal. The roster consumes it as a typed notification.
@@ -240,19 +243,6 @@ export function createCodexJournalTranslator(
}
}
/** Settles the item a notification the transport refused to carry left
* mid-flight; null when the frame is not one. */
function settleOversizedNotification(
event: Extract<CodexStructuredSessionEvent, { type: 'provider-frame' }>
): CodexJournalTranslationAdmission | null {
return settleCodexOversizedNotificationFrame({
...event,
sink: deps.sink,
streams: items.streams,
activeItems: items.activeItems
})
}
function startTurn(
event: Extract<CodexStructuredSessionEvent, { type: 'notification' }>
): CodexJournalTranslationAdmission {
@@ -8,6 +8,8 @@ import {
closeFailedCodexAcquisition,
stopSupersededCodexAcquisition
} from './codex-structured-acquisition-lifecycle'
import { CodexBackgroundTaskTracker } from './codex-background-task-tracker'
import { CodexSubagentExecutions } from './codex-subagent-executions'
import { createCodexJournalTranslator } from './codex-structured-journal-translation'
import { openCodexAppServerConnection } from './codex-app-server-connection'
import { codexProcessIdentity, codexProviderHandleLink } from './codex-structured-owner-identity'
@@ -74,10 +76,12 @@ export async function acquireCodexStructuredSession(input: {
acquireInput.identity.providerHandle.kind === 'codex'
? acquireInput.identity.providerHandle.threadId
: null
const subagentExecutions = new CodexSubagentExecutions()
const translator = acquireInput.events
? createCodexJournalTranslator({
sink: acquireInput.events,
primaryThreadId: () => primaryThreadId,
subagentExecutions,
bindPromptItemId: (journalItemId, threadId, promptKey) =>
acquisition.prompts.bindJournalItemId(journalItemId, threadId, promptKey)
})
@@ -138,6 +142,7 @@ export async function acquireCodexStructuredSession(input: {
connection: acquisition.connection,
error,
prompts: acquisition.prompts,
onBackgroundTasksChanged: deps.onBackgroundTasksChanged,
...(deps.onEvent ? { onEvent: deps.onEvent } : {})
})
} finally {
@@ -199,6 +204,7 @@ export async function acquireCodexStructuredSession(input: {
reportedOptions: reportedCodexThreadOptions(opened),
turnIdWaiters: [],
translator,
backgroundTasks: new CodexBackgroundTaskTracker(opened.threadId, subagentExecutions),
forceCloseUnexpected: (reason) =>
input.forceCloseUnexpected(
sessionId,
@@ -16,11 +16,7 @@ import type { CodexJournalTranslationAdmission } from './codex-structured-journa
import { answerCodexPrompt } from './codex-structured-prompt-replies'
import { dispatchCodexTurn, isCodexTurnOptionKey } from './codex-structured-turn-start'
import { supportsCodexStructuredLocation } from './codex-structured-location-support'
import {
closeAllCodexSessions,
closeCodexPublishedSession,
closeCodexSession
} from './codex-structured-session-close'
import { CodexStructuredSessionTeardown } from './codex-structured-session-teardown'
import {
applyCodexStructuredSessionOption,
readLiveCodexSessionOptions
@@ -54,6 +50,7 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap
private readonly acquisitions = new CodexAcquisitionRegistry()
private readonly turnCancellation: CodexStructuredTurnCancellation
private readonly notificationRetries: ReturnType<typeof createCodexStructuredNotificationRetry>
private readonly teardown: CodexStructuredSessionTeardown
constructor(private readonly deps: CodexStructuredSessionAdapterDeps) {
this.notificationRetries = createCodexStructuredNotificationRetry({
@@ -61,6 +58,15 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap
translate: (sessionId, session, method, params) =>
this.translateNotification(sessionId, session, method, params)
})
this.teardown = new CodexStructuredSessionTeardown({
sessions: this.sessions,
acquisitions: this.acquisitions,
...(deps.onEvent ? { onEvent: deps.onEvent } : {}),
...(deps.onBackgroundTasksChanged
? { onBackgroundTasksChanged: deps.onBackgroundTasksChanged }
: {}),
forgetNotificationRetries: (sessionId) => this.notificationRetries.clear(sessionId, null)
})
this.turnCancellation = new CodexStructuredTurnCancellation({
captureTurnProcesses: deps.captureTurnProcesses,
terminateTurnProcesses: deps.terminateTurnProcesses,
@@ -92,7 +98,7 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap
handleUnhandledFrame: (sessionId, kind, payload) =>
this.handleUnhandledFrame(sessionId, kind, payload),
forceCloseUnexpected: (sessionId, fence, acquisitionGeneration, reason) =>
this.forceCloseUnexpected(sessionId, fence, acquisitionGeneration, reason)
this.teardown.forceCloseUnexpected(sessionId, fence, acquisitionGeneration, reason)
})
/** Buffers pre-publication events and drops events from superseded children. */
@@ -134,12 +140,20 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap
session: CodexSession,
event: CodexStructuredSessionEvent
): CodexJournalTranslationAdmission {
if (event.type === 'notification' && !session.backgroundTasks.canObserve(event)) {
return { accepted: false, reason: 'failed' }
}
const admission = session.translator?.handle(event) ?? { accepted: true }
if (!admission.accepted) {
return admission
}
if (event.type === 'notification') {
this.compactions.codex(event.sessionId, event.method, event.params)
// After the admission check, so a refused frame is observed by the strip
// only on the retry that also reaches the journal.
if (session.backgroundTasks.observe(event)) {
this.deps.onBackgroundTasksChanged?.(event.sessionId, session.backgroundTasks.state)
}
}
if (event.type === 'ended') {
this.compactions.ended(event.sessionId)
@@ -167,6 +181,10 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap
)
}
backgroundTaskState: NonNullable<StructuredAgentSessionAdapter['backgroundTaskState']> = (
sessionId
) => this.sessions.get(sessionId)?.backgroundTasks.state
bindPromptItemId = (sessionId: string, journalItemId: string, promptKey: string): void =>
this.sessions
.get(sessionId)
@@ -267,59 +285,12 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap
identity: AgentSessionJournalIdentity
}): Promise<string | null> => this.sessions.get(input.identity.sessionId)?.historyPath ?? null
closeSession = async (sessionId: string): Promise<boolean> => {
const closed = await closeCodexSession(
sessionId,
this.sessions,
this.acquisitions,
this.deps.onEvent
)
if (closed) {
this.notificationRetries.clear(sessionId, null)
}
return closed
}
forceCloseSession = async (sessionId: string): Promise<boolean> => {
const closed = await closeCodexPublishedSession(this.sessions, sessionId, this.deps.onEvent, {
allowFailedSettlement: true,
requestedClose: false
})
if (closed) {
this.notificationRetries.clear(sessionId, null)
}
return closed
}
private forceCloseUnexpected(
sessionId: string,
fence: number,
acquisitionGeneration: string,
reason: Error
): Promise<boolean> {
const session = this.sessions.get(sessionId)
if (
!session ||
session.ended ||
session.fence !== fence ||
session.acquisitionGeneration !== acquisitionGeneration
) {
return Promise.resolve(false)
}
return closeCodexPublishedSession(this.sessions, sessionId, this.deps.onEvent, {
allowFailedSettlement: true,
requestedClose: false,
expectedFence: fence,
expectedAcquisitionGeneration: acquisitionGeneration,
unexpectedReason: reason
})
}
disposeSession = (sessionId: string): Promise<boolean> => this.closeSession(sessionId)
closeAll = (): Promise<void> =>
closeAllCodexSessions(this.sessions, this.acquisitions, (sessionId) =>
this.disposeSession(sessionId)
)
closeSession = (sessionId: string): Promise<boolean> => this.teardown.close(sessionId)
forceCloseSession = (sessionId: string): Promise<boolean> => this.teardown.forceClose(sessionId)
disposeSession = (sessionId: string): Promise<boolean> => this.teardown.close(sessionId)
closeAll = (): Promise<void> => this.teardown.closeAll()
releaseAcquisition = (input: { sessionId: string }): Promise<boolean> =>
this.closeSession(input.sessionId)
this.teardown.close(input.sessionId)
private session(sessionId: string): CodexSession {
return requireLiveCodexSession(this.sessions, sessionId)
@@ -0,0 +1,258 @@
import { describe, expect, it, vi } from 'vitest'
import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types'
import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire'
import type {
CodexAppServerConnection,
CodexAppServerConnectionHandlers,
openCodexAppServerConnection
} from './codex-app-server-connection'
import { CodexStructuredSessionAdapter } from './codex-structured-session-adapter'
import { CodexBackgroundTaskTracker } from './codex-background-task-tracker'
import type { CodexStructuredSessionEvent } from './codex-structured-session-state'
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
// Proves the strip is actually REACHED from provider traffic: the tracker is
// unit-tested separately, and a producer that is correct but unwired publishes
// nothing while every one of its own tests stays green.
const THREAD_ID = '01a07d54-3785-71d0-b065-82c8ebbc572a'
const PARENT_TURN = '01a07d54-37be-72e1-8206-8f0c23dd2cef'
const CHILD_ID = '01a07d54-5523-78a3-91f5-e0acb1dab065'
/** A three-route stand-in, deliberately smaller than the full adapter harness:
* this suite only needs a thread and a notification pipe. */
function fakeCodex(close: () => Promise<boolean> = async () => true): {
handlers: () => CodexAppServerConnectionHandlers
openConnection: typeof openCodexAppServerConnection
} {
let live: CodexAppServerConnectionHandlers = {}
const openConnection = (async (_launch, handlers = {}) => {
live = handlers
const connection: CodexAppServerConnection = {
pid: 4321,
closed: false,
request: async (method) =>
method === 'thread/start' ? { thread: { id: THREAD_ID, path: null } } : {},
notify: () => {},
respond: () => {},
respondWithError: () => {},
close
} as unknown as CodexAppServerConnection
return connection
}) as typeof openCodexAppServerConnection
return { handlers: () => live, openConnection }
}
function identity(sessionId: string): AgentSessionJournalIdentity {
return {
sessionId,
workspaceId: 'ws-1',
hostId: 'host-1',
agent: 'codex',
providerHandle: { kind: 'codex', threadId: THREAD_ID }
}
}
function subagentNotification(kind: string): { method: string; params: unknown } {
return {
method: 'item/started',
params: {
item: {
type: 'subAgentActivity',
id: 'call_1',
kind,
agentThreadId: CHILD_ID,
agentPath: '/root/count_a'
},
threadId: THREAD_ID,
turnId: PARENT_TURN
}
}
}
const TURN_COMPLETED = {
method: 'turn/completed',
params: { threadId: THREAD_ID, turn: { id: PARENT_TURN, status: 'completed' } }
}
async function adapterWithSession(
published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[],
events?: StructuredAgentSessionEventSink,
onEvent?: (event: CodexStructuredSessionEvent) => void,
close?: () => Promise<boolean>
): Promise<{ adapter: CodexStructuredSessionAdapter; codex: ReturnType<typeof fakeCodex> }> {
const codex = fakeCodex(close)
const adapter = new CodexStructuredSessionAdapter({
resolveLaunch: async () => ({
command: 'codex',
args: ['app-server'],
cwd: '/work/repo',
codexHome: null,
resumeThreadId: null
}),
openConnection: codex.openConnection,
readProcessStartTime: async () => 1_700_000_000_000,
onEvent,
onBackgroundTasksChanged: (sessionId, state) => published.push({ sessionId, state })
})
await adapter.acquire({
identity: identity('session-1'),
fence: 7,
spawnToken: 'spawn-9',
events
})
codex.handlers().onNotification?.('turn/started', {
threadId: THREAD_ID,
turn: { id: PARENT_TURN, status: 'inProgress' }
})
codex.handlers().onNotification?.('turn/started', {
threadId: CHILD_ID,
turn: { id: 'child-turn', status: 'inProgress' }
})
return { adapter, codex }
}
describe('codex background tasks reach the strip', () => {
it('clears natural-exit state before lifecycle observers can read it', async () => {
const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = []
const onEvent = vi.fn()
const { adapter, codex } = await adapterWithSession(published, undefined, onEvent)
const spawn = subagentNotification('started')
codex.handlers().onNotification?.(spawn.method, spawn.params)
codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params)
expect(adapter.backgroundTaskState('session-1')?.tasks).toHaveLength(1)
published.length = 0
onEvent.mockImplementation((event: CodexStructuredSessionEvent) => {
if (event.type === 'ended') {
expect(adapter.backgroundTaskState('session-1')).toBeNull()
}
})
codex.handlers().onExit?.(new Error('provider exited'))
expect(adapter.backgroundTaskState('session-1')).toBeNull()
expect(published).toEqual([{ sessionId: 'session-1', state: null }])
await adapter.closeSession('session-1')
})
it('keeps live tasks when close is refused', async () => {
const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = []
const close = vi.fn(async () => false)
const { adapter, codex } = await adapterWithSession(published, undefined, undefined, close)
const spawn = subagentNotification('started')
codex.handlers().onNotification?.(spawn.method, spawn.params)
codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params)
const before = adapter.backgroundTaskState('session-1')
published.length = 0
expect(await adapter.closeSession('session-1')).toBe(false)
expect(adapter.backgroundTaskState('session-1')).toEqual(before)
expect(published).toEqual([])
close.mockResolvedValue(true)
await adapter.closeSession('session-1')
})
it('does not let an old exit callback clear a replacement roster', async () => {
const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = []
const { adapter, codex } = await adapterWithSession(published)
const oldExit = codex.handlers().onExit
await adapter.acquire({ identity: identity('session-1'), fence: 8, spawnToken: 'spawn-10' })
codex.handlers().onNotification?.('turn/started', {
threadId: CHILD_ID,
turn: { id: 'replacement-child-turn' }
})
const spawn = subagentNotification('started')
codex.handlers().onNotification?.(spawn.method, spawn.params)
const before = adapter.backgroundTaskState('session-1')
expect(before?.tasks).toHaveLength(1)
published.length = 0
oldExit?.(new Error('old provider exited late'))
expect(adapter.backgroundTaskState('session-1')).toEqual(before)
expect(published).toEqual([])
await adapter.closeSession('session-1')
})
it('recovers the exact provider generation when command metadata cannot be admitted', async () => {
const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = []
const observed: CodexStructuredSessionEvent[] = []
const appendItem = vi.fn()
const { adapter, codex } = await adapterWithSession(
published,
{ appendItem, appendTombstone: () => {}, publish: () => {} },
(event) => observed.push(event)
)
appendItem.mockClear()
observed.length = 0
const admission = vi
.spyOn(CodexBackgroundTaskTracker.prototype, 'canObserve')
.mockReturnValue(false)
try {
codex.handlers().onNotification?.('item/started', {
threadId: THREAD_ID,
turnId: PARENT_TURN,
item: {
type: 'commandExecution',
id: 'over-budget',
command: 'sleep 1',
source: 'unifiedExecStartup',
status: 'inProgress'
}
})
await vi.waitFor(() => expect(adapter.backgroundTaskState('session-1')).toBeUndefined())
expect(appendItem.mock.calls.map((call) => call[1])).toEqual([
{ kind: 'status', text: 'Provider exited: notification admission failed (failed)' }
])
expect(observed).toEqual([
expect.objectContaining({
type: 'ended',
cause: 'unexpected-exit',
fence: 7,
acquisitionGeneration: expect.any(String),
reason: 'notification admission failed (failed)'
})
])
expect(published).toEqual([{ sessionId: 'session-1', state: null }])
} finally {
admission.mockRestore()
await adapter.closeSession('session-1')
}
})
it('publishes the orphaned fan-out once the spawning turn completes', async () => {
const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = []
const { adapter, codex } = await adapterWithSession(published)
const spawn = subagentNotification('started')
codex.handlers().onNotification?.(spawn.method, spawn.params)
// The child is still inside the turn, so the strip stays silent.
expect(published).toEqual([])
expect(adapter.backgroundTaskState('session-1')).toBeNull()
codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params)
expect(published).toEqual([
{
sessionId: 'session-1',
state: {
state: 'monitoring',
supportsStopAll: false,
tasks: [{ id: `codex-agent:${CHILD_ID}`, kind: 'agent', description: 'count_a' }]
}
}
])
expect(adapter.backgroundTaskState('session-1')).toEqual(published[0].state)
})
it('clears the strip when the session closes', async () => {
const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = []
const { adapter, codex } = await adapterWithSession(published)
const spawn = subagentNotification('started')
codex.handlers().onNotification?.(spawn.method, spawn.params)
codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params)
published.length = 0
expect(await adapter.closeSession('session-1')).toBe(true)
// Explicit null, not silence: the reader answers `undefined` once the
// session is gone, which every channel treats as "unchanged".
expect(published).toEqual([{ sessionId: 'session-1', state: null }])
expect(adapter.backgroundTaskState('session-1')).toBeUndefined()
})
})
@@ -10,6 +10,7 @@ import {
type CodexStructuredSessionEvent
} from './codex-structured-session-adapter'
import { handleCodexSessionExit } from './codex-structured-session-close'
import { CodexBackgroundTaskTracker } from './codex-background-task-tracker'
import type { CodexSession } from './codex-structured-session-state'
import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter'
import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router'
@@ -90,6 +91,7 @@ describe('Codex structured session close lifecycle', () => {
} as unknown as NonNullable<CodexSession['translator']>
const session = {
connection,
backgroundTasks: new CodexBackgroundTaskTracker('thread-1'),
ended: false,
requestedClose: false,
fence: 7,
@@ -4,6 +4,7 @@ import {
cancelCodexAcquisitionAttempt,
type CodexAcquisitionRegistry,
type CodexSession,
type CodexStructuredSessionAdapterDeps,
type CodexStructuredSessionEvent
} from './codex-structured-session-state'
import type { StructuredAgentSessionLifecycleEvent } from '../native-chat/agent-session-wire/structured-agent-session-adapter'
@@ -16,6 +17,7 @@ export function handleCodexSessionExit(input: {
prompts?: CodexSession['prompts']
allowFailedSettlement?: boolean
onEvent?: (event: CodexStructuredSessionEvent) => void
onBackgroundTasksChanged?: CodexStructuredSessionAdapterDeps['onBackgroundTasksChanged']
}): boolean {
const session = input.sessions.get(input.sessionId)
if (!session || session.connection !== input.connection || session.ended) {
@@ -43,6 +45,8 @@ export function handleCodexSessionExit(input: {
event.settlementRetryRequired = true
}
session.ended = true
session.backgroundTasks.clear()
input.onBackgroundTasksChanged?.(input.sessionId, null)
session.unbindReadingControl?.()
input.onEvent?.(event)
session.prompts.clear()

Some files were not shown because too many files have changed in this diff Show More