diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index f75d7ba00bb..942f0f34a56 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -227,6 +227,7 @@ jobs: mapfile -t TEST_FILES < <(jq -r '.[] | select( . != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and . != "tests/e2e/paired-startup-exec-readiness.spec.ts" and + . != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts" and . != "tests/e2e/local-ssh-browser-routing.spec.ts" and . != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and . != "tests/e2e/ssh-localhost.spec.ts" and @@ -272,6 +273,7 @@ jobs: if: >- inputs.test_files == '' || inputs.ssh_source_changed == 'true' || + contains(inputs.test_files, 'tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts') || contains(inputs.test_files, 'tests/e2e/local-ssh-browser-routing.spec.ts') || contains(inputs.test_files, 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts') || contains(inputs.test_files, 'tests/e2e/ssh-startup-exec-readiness.spec.ts') || diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index b481170ec99..1c1c2c4edd1 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -17,15 +17,33 @@ "protection": "partial", "owner": "runtime", "layer": "service-integration", - "surfaces": ["headless startup", "push registration", "mobile notification replay"], - "platforms": ["macos", "linux", "windows"], - "providers": ["local", "remote-runtime"], - "coveredPlatforms": ["macos"], - "coveredProviders": ["local", "remote-runtime"], + "surfaces": [ + "headless startup", + "push registration", + "native push delivery policy" + ], + "platforms": [ + "macos", + "linux", + "windows" + ], + "providers": [ + "local", + "remote-runtime" + ], + "coveredPlatforms": [ + "macos" + ], + "coveredProviders": [ + "local", + "remote-runtime" + ], "coverageNotes": "Actual startOrcad entry with mocked daemon/RPC startup boundaries; real controller, push service, and persisted device registry. Gateway send is stubbed.", - "motivatingLinks": ["https://github.com/stablyai/orca/pull/19204"], - "invariant": "Headless startup installs and disposes push delivery; push and replay preserve the three-minute away policy, and host activity cannot extend the seven-day mobile lease.", - "oracle": "Require registration after RPC identity initialization and shutdown cleanup; idle 179/180/0 yields false/true/false for socket and replay, and exactly one push. Persisted lease expires exactly at seven days and only explicit registration renews it.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/pull/19204" + ], + "invariant": "Headless startup installs and disposes push delivery; native push is the only mobile OS-banner path, desktop notification categories remain authoritative, the three-minute away policy is preserved, and host activity cannot extend the seven-day mobile lease. The gateway emits individual notifications rather than custom summaries.", + "oracle": "Require registration after RPC identity initialization and shutdown cleanup; idle 179/180/0 yields false/true/false in retained event metadata and exactly one gateway push. Socket and reconnect paths retain app state and tray reconciliation without creating OS banners, desktop categories remain authoritative, and gateway alerts retain individual identities. Persisted lease expires exactly at seven days and only explicit registration renews it.", "commands": [ "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/mobile-notification-dismissal-store.test.ts src/renderer/src/hooks/useAutoAckViewedAgent.away.test.ts", "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/orcad/orcad-push-startup.test.ts src/main/runtime/push/push-policy-pipeline.integration.test.ts" @@ -46,7 +64,7 @@ { "file": "src/main/runtime/push/push-policy-pipeline.integration.test.ts", "assertions": [ - "carries the native idle boundary through replay, socket policy and push dispatch", + "carries the native idle boundary through replay and push dispatch", "expires persisted registration at seven days despite host activity and renews explicitly" ] } @@ -78,15 +96,172 @@ "required": false, "evidence": "Lifecycle and policy coverage; asserts exact gateway send counts and zero remaining dispatch listeners after shutdown." }, - "promotionCriteria": ["Collect repeated CI runs without unexplained failures."], + "promotionCriteria": [ + "Collect repeated CI runs without unexplained failures." + ], "knownGaps": [ "Does not prove APNs silent background wakeup or actual operating-system idle transitions.", "Rendererless agent/bell event generation remains outside the documented feature contract.", "No live Windows or Linux policy evidence.", - "Native iOS dismissal callback processing is verified separately; real APNs background wakeup is not established. Coalesced summaries require membership-aware dismissal before they can be cleared safely." + "Native iOS dismissal callback processing is verified separately; real APNs background wakeup and Android automatic grouping are not established. Legacy summaries already delivered during rolling overlap rely on the bounded compatibility reader outside this gate." ], "demotionRule": "Keep experimental if lifecycle or policy assertions fail; do not weaken them to bypass platform delivery gaps." }, + { + "id": "agent-session.history-forward-read-budget", + "title": "Journal catch-up reads only the next page and one lookahead row", + "maturity": "experimental", + "protection": "partial", + "owner": "agent-session-runtime", + "layer": "runtime-unit", + "surfaces": ["structured agent history", "structured agent subscriptions"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "ssh", "remote-runtime"], + "coverageNotes": "The real SQLite journal and production subscriber delivery are exercised with a folder workspace and remote host identity. The SQL and pagination code is shared across execution hosts; live SSH transport and Linux/Windows runtime execution are not exercised. PTY, daemon, WSL execution, and mobile rendering are unaffected.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts" + ], + "invariant": "Forward catch-up preserves every item, revision, tombstone, sequence cursor, page byte bound, and reset behavior while reading at most the requested row count plus one from SQLite for each page.", + "oracle": "Reconnect a real subscriber to a 2,000-row journal and receive all 2,000 item identities in order through the live cursor; count the actual SQL rows returned and parsed as 2,009 instead of 11,000. Assert exact final-page hasNewer, unlimited reader compatibility, gap detection at the next page, and parse-stop behavior at the lookahead row. Existing history tests cover revisions, tombstones, byte-bound shrinking, epochs, and schema resets.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts src/main/native-chat/agent-session-journal" + ], + "testFiles": [ + "src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts", + "src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts", + "src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts" + ], + "assertionRefs": [ + { + "file": "src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts", + "assertions": [ + "reconnects through every page with one lookahead row per page", + "keeps an exact final page final and preserves unlimited journal readers", + "reports a sequence gap when the next page reaches it", + "preserves parse-stop behavior at lookahead: %s" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts src/main/native-chat/agent-session-journal", + "result": "passed", + "durationSeconds": 9.94, + "summary": "214 tests passed across 19 files, including actual SQLite row and JSON parse counts through production subscriber catch-up." + } + ], + "runtimeBudget": { + "p95Seconds": 30, + "scope": "Real SQLite journal unit and production subscriber tests; no launched app." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial deterministic local validation; CI soak has not started." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "Before the change, SQL returned 2,000, 1,800, 1,600 through 200 rows across ten pages, failing the count assertion. The bounded query returns nine pages of 201 rows and a final 200, with exactly 2,009 row parses and identical item delivery." + }, + "performanceBudget": { + "required": true, + "evidence": "Catch-up materialization and JSON parsing are linear in unseen journal rows plus page lookaheads. A cached parameterized LIMIT adds no polling, cache invalidation, output loss, protocol change, or provider calls." + }, + "knownGaps": [ + "Linux and Windows execution and live SSH transport have not been exercised.", + "The existing full reduced-state snapshot and batch projection cost are outside this SQL read budget." + ], + "promotionCriteria": [ + "Complete CI soak requirements while preserving the deterministic row budget and pagination oracles." + ], + "demotionRule": "Keep experimental until CI soak; investigate fidelity or count failures without relaxing the row budget." + }, + { + "id": "terminal-performance.osc-status-scan-budget", + "title": "OSC 9999 status bursts reuse forward terminator searches", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "shared-unit-and-runtime-unit", + "surfaces": ["terminal output ingestion", "terminal agent-status side effects"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "daemon", "ssh", "remote-runtime"], + "coverageNotes": "Shared parser tests cover provider-independent bytes; main and renderer contract tests cover status and terminal-output delivery. Live Linux, Windows, WSL, SSH and remote-runtime processes are not launched. Execution, liveness, paths, wire formats and mobile UI are unchanged.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/src/shared/agent-status-osc.ts" + ], + "invariant": "Terminal status parsing preserves ordinary UTF-16 output, every valid payload in order, the last valid payload's clean-output offset, earliest BEL/ST termination, and incomplete-frame caps while searching each complete burst only forward.", + "oracle": "Two 5,000-frame bursts using exclusively BEL or ST produce every expected payload and ordinary output byte with at most twice the input length in native search ranges. Mixed terminators, every split through prefixes/JSON/ST, independent parser interleaving, malformed payloads, exact pending-cap boundaries and oversized complete frames retain their previous behavior. A one-character echo performs no terminator search. Parsed output chunks are not retained in legacy regular-expression state.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-scan-budget.test.ts src/shared/agent-status-types.test.ts src/main/runtime/orca-runtime-hook-agent-status-projection.test.ts src/renderer/src/components/terminal-pane/terminal-title-tracker-parity.test.ts src/renderer/src/components/terminal-pane/pty-connection-main-side-effect-authority.test.ts src/renderer/src/components/terminal-pane/pty-connection-hook-completion-side-effects.test.ts src/renderer/src/components/terminal-pane/pty-transport-eager-buffer-replay.test.ts" + ], + "testFiles": [ + "src/shared/agent-status-osc.test.ts", + "src/shared/agent-status-osc-scan-budget.test.ts", + "src/main/runtime/orca-runtime-hook-agent-status-projection.test.ts", + "src/renderer/src/components/terminal-pane/terminal-title-tracker-parity.test.ts" + ], + "assertionRefs": [ + { + "file": "src/shared/agent-status-osc-scan-budget.test.ts", + "assertions": [ + "reads each burst only forward with terminator %j", + "keeps a one-character input echo on the ordinary-output path", + "does not retain the output chunk in legacy regular-expression state" + ] + }, + { + "file": "src/shared/agent-status-osc.test.ts", + "assertions": [ + "uses the earliest mixed terminator and counts only parsed payload offsets", + "keeps a distant ST usable after many intervening BEL frames", + "preserves every split of prefixes, JSON, and both terminators across independent streams", + "applies the pending cap only to incomplete frames" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-scan-budget.test.ts src/shared/agent-status-types.test.ts src/main/runtime/orca-runtime-hook-agent-status-projection.test.ts src/renderer/src/components/terminal-pane/terminal-title-tracker-parity.test.ts src/renderer/src/components/terminal-pane/pty-connection-main-side-effect-authority.test.ts src/renderer/src/components/terminal-pane/pty-connection-hook-completion-side-effects.test.ts src/renderer/src/components/terminal-pane/pty-transport-eager-buffer-replay.test.ts", + "result": "passed", + "durationSeconds": 23.96, + "summary": "153 tests passed across eight files. Independent baseline differential review also matched 3,704 streams and 45,141 chunk results." + } + ], + "runtimeBudget": { + "p95Seconds": 30, + "scope": "Shared parser and main/renderer terminal contract tests; no launched app." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial deterministic local validation; CI soak has not started." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "The unchanged parser failed both search budgets: 618,560,785 searched characters for the 246,390-character BEL burst and 631,068,285 for the 251,390-character ST burst. Reusing forward match positions reduces those totals to 492,770 and 502,770 characters respectively, within twice the input length, with identical complete results." + }, + "performanceBudget": { + "required": true, + "evidence": "Warmed Node 24 macOS CPU medians: a 250 KB / 5,000-status burst fell from 100.240 ms to 1.903 ms; a 1 MB / 20,000-status burst fell from 1,588.492 ms to 7.369 ms. Wall-clock medians were 134.878 to 2.484 ms and 2,536.927 to 12.902 ms under concurrent machine load. These are adverse bursts, not typical callback sizes. The ordinary-output path is unchanged; one-character echo CPU was 2.173 versus 2.342 ms per 100,000 calls, and single-status BEL CPU was 10.835 versus 10.897 ms per 30,000 calls. Both native terminator searches advance monotonically within the current chunk; no regex state retains the input. No scheduling, polling, provider calls, pending limits, output filtering or payload parsing changed." + }, + "knownGaps": [ + "Live Electron input latency and Linux/Windows/WSL/SSH execution have not been measured for this parser-only change.", + "Fragmented unterminated payload accumulation and downstream processing of large status arrays remain outside this complete-burst search budget." + ], + "promotionCriteria": [ + "Complete CI soak while preserving byte fidelity and deterministic search budgets." + ], + "demotionRule": "Keep experimental until CI soak; investigate output, offset, carry or search-budget failures without relaxing the oracle." + }, { "id": "terminal-performance.padded-fullscreen-redraw", "title": "Fullscreen redraw padding does not stall terminal delivery", @@ -345,6 +520,106 @@ ], "demotionRule": "Keep experimental or demote if adoption duplicates covered output, drops newer or unproven output, changes terminal ownership, or flakes without explanation." }, + { + "id": "terminal-session.io-failure-cleanup", + "title": "Native PTY I/O failures preserve termination ownership", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "provider-contract", + "surfaces": ["daemon PTY teardown"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local-daemon", "ssh-daemon", "paired-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local-daemon"], + "coverageNotes": "Real TerminalHost, Session, and subprocess wrapper with injected native I/O failures and mocked OS signals. Local non-daemon and SSH-relay implementations are unaffected; daemon consumers on SSH, WSL, paired runtimes, and mobile retain host-owned semantics. Live Linux/Windows/WSL and remote runs remain gaps. No git or folder-workspace assumptions. The fault-injection suite also runs with simulated darwin/linux/win32 platform branches; these do not constitute native OS coverage. Native macOS coverage now proves shell exit and PTY master-fd closure, input/output round trips, and teardown of a paused producer for both graceful and immediate cleanup. Windows single-close/job escalation and pre-listener output/status are fault-injected contracts.", + "motivatingLinks": ["docs/terminal-daemon-session-leak-investigation.md"], + "invariant": "I/O errors must not establish physical exit or disable termination of an owned PTY. Session and native handle disposal require the exit event.", + "oracle": "Inject write and resize failures, require graceful and forced signals to reach the native owner, keep producer resume available, suppress repeated failed I/O, deliver output and exit, and suppress signals after exit. Across 32 create/close cycles per failure, retain each session before exit and release its native handle and emulator exactly once afterwards. Mark physical exit before notifying listeners; reentrant kill/forceKill/signal from those listeners must never signal the retired PID. A native POSIX test performs input/output and resize, pauses the producer, injects each I/O failure, then gracefully or immediately closes 16 real shells; require ESRCH for each child PID and EBADF for each PTY master fd.", + "commands": [ + "pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts", + "pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts src/main/daemon/pty-subprocess-handle-lifecycle.test.ts src/main/daemon/terminal-host-session-reaping-leak.test.ts src/main/daemon/terminal-host-teardown-recreate.test.ts src/main/daemon/terminal-session-teardown.test.ts src/main/daemon/session.test.ts", + "pnpm test src/main/daemon/pty-subprocess-io-failure-native.test.ts" + ], + "testFiles": [ + "src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts", + "src/main/daemon/pty-subprocess-handle-lifecycle.test.ts", + "src/main/daemon/terminal-host-session-reaping-leak.test.ts", + "src/main/daemon/terminal-host-teardown-recreate.test.ts", + "src/main/daemon/terminal-session-teardown.test.ts", + "src/main/daemon/session.test.ts", + "src/main/daemon/pty-subprocess-io-failure-native.test.ts" + ], + "assertionRefs": [ + { + "file": "src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts", + "assertions": [ + "keeps graceful and forced termination available until physical exit", + "reaps every session and native handle across 32 failed-I/O create/close cycles", + "suppresses repeated native I/O failures while still delivering output and exit", + "blocks reentrant termination from an exit listener after I/O failure" + ] + }, + { + "file": "src/main/daemon/pty-subprocess-io-failure-native.test.ts", + "assertions": ["reaps real shells and master fds after %s failure (immediate=%s)"] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts src/main/daemon/pty-subprocess-handle-lifecycle.test.ts src/main/daemon/terminal-host-session-reaping-leak.test.ts src/main/daemon/terminal-host-teardown-recreate.test.ts src/main/daemon/terminal-session-teardown.test.ts src/main/daemon/session.test.ts", + "result": "passed", + "durationSeconds": 0.617, + "summary": "136 tests passed across six files; failed-I/O cycle tests cover 64 closures." + }, + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts", + "result": "passed", + "durationSeconds": 3.71, + "summary": "32 tests passed with 4 Windows-only cases skipped; simulated macOS/Linux/Windows branches include 192 failed-I/O create/close cycles and exit-listener reentrancy." + }, + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/main/daemon/pty-subprocess-io-failure-native.test.ts", + "result": "passed", + "durationSeconds": 2.52, + "summary": "Four native cases pass across 16 real shells, including input/output, pause before teardown, confirmed PID absence, and closed PTY master fds." + } + ], + "runtimeBudget": { + "p95Seconds": 10, + "scope": "focused daemon teardown contract tests" + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial deterministic local run; no soak history." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "All four original regression cases failed before the fix because native kill was never called; the unchanged cases passed after separating I/O failure from exit. Two additional output/flow-control cases also pass. Review added two failing exit-listener reentrancy cases; publishing physical exit before callbacks made them pass." + }, + "performanceBudget": { + "required": true, + "evidence": "One boolean per PTY; no new timers, scans, retries, or subprocesses. Existing failed-I/O suppression remains. 192 closures under three simulated platform branches return session inventory to zero and dispose each emulator/native handle once." + }, + "promotionCriteria": [ + "Collect remaining native cross-platform evidence plus the standard soak history." + ], + "knownGaps": [ + "Fault injection proves a leak mechanism, not causality for the historical 427-session incident.", + "Real Linux/Windows/WSL, remote, startup-close, login-wrapper descendants, and multi-day load evidence remain outstanding. Native tests inject synchronous I/O errors; they do not model every asynchronous node-pty pipe failure.", + "No output throughput change or interactive latency benchmark is included." + ], + "demotionRule": "Keep experimental; investigate any lost cleanup signal, premature exit, or unexplained flake." + }, { "id": "cmd-j-tabs.host-qualified-candidate-ownership", "title": "Cmd-J tab candidates retain execution-host ownership", @@ -5902,7 +6177,7 @@ "invariant": "After a TUI exits or is killed, reveal, reattach, snapshot replay, or renderer remount must not deliver terminal-owned mouse or alternate-screen protocol bytes to the surviving shell. Recovery is an ordered output barrier in the daemon session data path: an OSC 133;D completing while the alternate screen is still active pauses the stream at that exact byte boundary, a fresh execution-host process inspection proves shell ownership, and on proof a mode reset is injected as in-stream output so every consumer converges by parsing the same bytes and the queued post-boundary shell output (the prompt) lands on the normal buffer. Snapshots are pure reads. Any failure — refuted proof, timeout, queue overflow, session death, disposal — flushes the queue unmodified, preserving incumbent behavior; later command or mode bytes revoke proof. Clean alternate-screen exits prove ownership asynchronously without pausing.", "oracle": "Run one fixed child-TUI journey for normal exit and cleanup-free SIGKILL. Assert renderer and host normal-buffer/non-mouse state, host snapshot terminalOwner metadata, exact PTY writes with no post-exit mouse report, unrelated-pane survival, post-boundary prompt output preserved (normal exit), ordered proof invalidation, bounded settlement and bail-out flush, one inspection per unclean episode with zero scans for ordinary output, split-escape safety at every chunk boundary, and old/new client-host fallback parity.", "commands": [ - "pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/terminal-shell-lifecycle-scanner.test.ts src/main/daemon/terminal-shell-recovery-barrier.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/session.test.ts src/main/daemon/terminal-host-concurrent-create.test.ts src/main/daemon/daemon-pty-adapter.test.ts src/main/daemon/daemon-restore-scrollback-depth.test.ts src/main/daemon/terminal-checkpoint-serializer.test.ts src/main/providers/agent-foreground-process.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/mobile-subscribe-integration.test.ts src/main/runtime/rpc/terminal-multiplex-escape-tail.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-codex-queries.test.ts src/renderer/src/components/terminal-pane/pty-connection-reattach-mode-reset.test.ts src/renderer/src/components/terminal-pane/pty-connection-daemon-snapshot-replay.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-escape-tail.test.ts --reporter=dot", + "pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/terminal-shell-lifecycle-scanner.test.ts src/main/daemon/terminal-shell-recovery-barrier.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/session.test.ts src/main/daemon/terminal-host-concurrent-create.test.ts src/main/daemon/daemon-pty-adapter.test.ts src/main/daemon/daemon-restore-scrollback-depth.test.ts src/main/daemon/terminal-checkpoint-serializer.test.ts src/main/providers/agent-foreground-process.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/mobile-subscribe-integration.test.ts src/main/runtime/rpc/terminal-multiplex-escape-tail.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-codex-queries.test.ts src/renderer/src/components/terminal-pane/pty-connection-reattach-mode-reset.test.ts src/renderer/src/components/terminal-pane/pty-connection-daemon-snapshot-replay.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-escape-tail.test.ts src/shared/terminal-partial-escape-tail.test.ts src/shared/terminal-partial-escape-tail.fuzz.test.ts --reporter=dot", "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/cross-version-wire/cross-version-terminal-wire.unit.test.ts --reporter=dot", "pnpm exec electron-vite build --mode e2e", "SKIP_BUILD=1 pnpm exec playwright test tests/e2e/terminal-hidden-child-tui-kill-mode-reset.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1" @@ -5924,6 +6199,8 @@ "src/renderer/src/components/terminal-pane/pty-connection-reattach-mode-reset.test.ts", "src/renderer/src/components/terminal-pane/pty-connection-daemon-snapshot-replay.test.ts", "src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-escape-tail.test.ts", + "src/shared/terminal-partial-escape-tail.test.ts", + "src/shared/terminal-partial-escape-tail.fuzz.test.ts", "tests/e2e/cross-version-wire/cross-version-terminal-wire.unit.test.ts", "tests/e2e/terminal-hidden-child-tui-kill-mode-reset.spec.ts" ], @@ -5945,6 +6222,13 @@ "a snapshot taken during a split escape keeps the pending tail intact and stale proof is revoked by the completing bytes" ] }, + { + "file": "src/shared/terminal-partial-escape-tail.fuzz.test.ts", + "assertions": [ + "the pending tail this gate threads over the wire folds identically at every code-unit split of the combined stream, including boundaries landing inside oscEsc/stringEsc", + "the split sweep runs over an alphabet carrying CAN, SUB, doubled ESC inside OSC/DCS/SOS/PM/APC, BEL, C1 ST, NUL, DEL, intermediates, CJK, astral, and lone surrogates" + ] + }, { "file": "tests/e2e/terminal-hidden-child-tui-kill-mode-reset.spec.ts", "assertions": [ @@ -13102,7 +13386,7 @@ "https://github.com/stablyai/orca/issues/13821", "https://github.com/stablyai/orca/issues/14347" ], - "invariant": "Injected orchestration task prompts for recognized agent CLIs must send the prompt body inside one bracketed-paste frame, sanitize embedded ESC bytes, preserve chunk boundaries without losing the frame, and submit exactly once only after the agent can accept Enter. A successful orchestration.workerStart must durably record exactly one accepted and started turn; a swallowed Enter must fail with agent_prompt_stalled and never trigger a blind rescue Enter. Claude and Codex must emit a post-paste composer marker and then settle, or reach the bounded fallback first; every other agent retains the platform delay.", + "invariant": "Injected orchestration task prompts for recognized agent CLIs must send the prompt body inside one bracketed-paste frame, sanitize embedded ESC bytes, preserve chunk boundaries without losing the frame, and submit exactly once only after the agent can accept Enter. Local worker-start with supported observation must preserve an unobserved turn as start_unknown without revoking authority, closing questions, or triggering a rescue Enter; a worker report during observation must settle normally. Claude and Codex must emit a post-paste composer marker and then settle, or reach the bounded fallback first; every other agent retains the platform delay.", "oracle": "Runtime tests assert the exact PTY write sequence, failure cleanup, Claude/Codex marker-gated multi-frame renders, and the legacy platform delay for every other configured agent. The candidate resets settlement on later frames, gives a late marker a fresh bounded window, and still submits once at the hard deadline if output never settles. The worker-start contract drives the production RPC through a delayed fake Codex composer and independently checks exact turn/Enter counts plus reopened SQLite Task, Dispatch, worker receipt, and mutation receipt state for accepted and swallowed outcomes. Other orchestration tests assert dispatch/coordinator use the agent prompt path; the live CLI harness covers long Codex-like framing.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts", @@ -13152,7 +13436,8 @@ "file": "src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts", "assertions": [ "delayed composer readiness produces exactly one submitted and started turn with no premature Enter and durable ready receipts", - "a swallowed Enter records agent_prompt_stalled across Task, Dispatch, worker, and mutation receipts without a rescue Enter" + "a swallowed Enter durably records start_unknown without a rescue Enter or capability revocation", + "early worker reports settle during observation, and outstanding questions survive observation uncertainty" ] }, { @@ -13553,6 +13838,7 @@ "invariant": "Starting a worker in the coordinator's current workspace must materialize one inactive terminal tab before worker-start returns, preserve coordinator focus, and remain exactly once after workspace re-entry. After an app update or restart, an exact live legacy worker must fence automatic provider resume, adopt its original PTY into its original background pane, retain readable output, and clear the resume record without spawning, writing, signalling, interrupting, replacing, or focusing the worker. A current-contract worker whose renderer graph identity is temporarily absent must retain its Dispatch capability and settle exactly once from exact hook-attested handle, pane, and process evidence; otherwise only an exact attested coordinator may take over. A worker_done caller may report success only after the owning runtime returns an explicit lifecycle verdict or authoritative reads prove that the exact Task, Dispatch, and worker report receipt settled the expected outcome. Federated terminal settlement must remain replay-eligible until the worker durably acknowledges it, and identical same-outcome retries must converge idempotently. Independently updated clients and worker servers must preserve the negotiated protocol: current peers use Run-home lifecycle settlement, while protocol v1/v2 peers retain their legacy completion path without receiving newer-only fields. A federated worker may accept only the authority defined by its negotiated protocol. An exact existing target workspace must receive a discoverable tab without stealing coordinator focus; if renderer reveal fails, worker-start must expose that the live worker remains background-only. Run and Dispatch checks must resolve through the caller's stable pane identity when a terminal handle is reminted, while a live handle outranks mismatched pane metadata. A nested worker's creator edge requires the current creator pane, process incarnation, and owning Run generation; reminting and rebinding that pane to another Run must remove the stale edge. Explicit legacy terminal inspection remains handle-scoped, and remote or headless worker presentation remains background-only.", "oracle": "Drive Run create, Task create, and worker-start through production Electron runtimes with a deterministic Codex fixture. Require append-only ledgers with one still-live PID and no interruption, a visible inactive worker tab while the coordinator stays active, Run delivery through stable pane identity, and stable PTY/incarnation, tab, leaf, worktree, Task, and Dispatch across workspace re-entry. In a restart journey, retain the original daemon PTY and PID, remove renderer ownership, retain sleeping-session evidence, mark the Dispatch legacy, relaunch, and require exact inactive tab adoption, readable ACK output, cleared resume state, one spawn, and no resume argv or Conversation interrupted text after another workspace round trip. The service oracle removes renderer lookup identity from current-contract callers while retaining real restored-PTY and hook commitments, replays authenticated completion and takeover across fresh runtimes, and requires one Task, Dispatch, terminal authority, message, mutation, ordinary-mail delivery, remote process fencing, and unchanged fixture marker bytes while foreign pane evidence remains rejected. Unit tests separately remint a creator pane and process from Run A into Run B, require the nested Run A worker to fall back to its current coordinator, require indexed query plans, and bound 300 Task reads with 50,000 retained Runs. They also assert authority-specific legacy affordances, exact identity and owner matching, retained-output fallback, pane-stable routing, federated non-activation, and SSH fallback parity.", "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 npx vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts --reporter=dot", "pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration.test.ts src/cli/handlers/orchestration-check-identity.test.ts src/cli/handlers/orchestration-worker-cli.test.ts src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts src/main/ssh/ssh-remote-orca-cli.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration-lifecycle-rejection.test.ts src/cli/handlers/orchestration-lifecycle-json-rejection.test.ts src/cli/handlers/orchestration-migration.test.ts", @@ -13567,6 +13853,7 @@ "pnpm run build:cli && SKIP_BUILD=1 pnpm exec playwright test tests/e2e/orchestration-worker-settlement-release-cli.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1" ], "testFiles": [ + "src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts", "src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts", "src/main/runtime/orchestration/formatter.test.ts", "src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts", @@ -13590,6 +13877,13 @@ "tests/e2e/orchestration-worker-settlement-release-cli.spec.ts" ], "assertionRefs": [ + { + "file": "src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts", + "assertions": [ + "replays the coordinator instruction and takes its ack after the app restarts", + "files loopback mail once under the local Dispatch Run without replacing its owner" + ] + }, { "file": "src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts", "assertions": [ @@ -14088,7 +14382,7 @@ "providers": ["local", "daemon", "ssh", "wsl", "remote-runtime"], "coveredPlatforms": ["macos"], "coveredProviders": ["local", "ssh"], - "coverageNotes": "Deterministic service tests cover release-versus-reuse ordering, transactional retain and takeover cancellation, exact host/pane/process identity, dead external/user-owned/transferred/stopped/abandoned reconciliation, host-partition persistence and legacy retirement replay with an absent web-terminal layout map, conservative unknown provider and legacy metadata handling, immutable transcript and bounded terminal archives, mutation restart, reset cleanup, replay idempotency, and 50-resource accounting. A macOS Electron journey invokes the freshly compiled worker-release CLI after the worker process disappears, then independently checks released SQLite state and coordinator liveness. Injected inventories cover local and SSH provider routing; live SSH, WSL, Windows, paired-runtime, and provider-close lost-ack journeys remain explicit gaps.", + "coverageNotes": "New phones explicitly report terminal takeover on real user sends, throttled per owning client and handle. Host byte lanes perform zero orchestration SQL work; local and injected SSH report tests fence release. Phones predating this build do not fence release. Deterministic service tests cover release-versus-reuse ordering, transactional retain and takeover cancellation, exact host/pane/process identity, dead external/user-owned/transferred/stopped/abandoned reconciliation, host-partition persistence and legacy retirement replay with an absent web-terminal layout map, conservative unknown provider and legacy metadata handling, immutable transcript and bounded terminal archives, mutation restart, reset cleanup, replay idempotency, and 50-resource accounting. A macOS Electron journey invokes the freshly compiled worker-release CLI after the worker process disappears, then independently checks released SQLite state and coordinator liveness. Injected inventories cover local and SSH provider routing; live SSH, WSL, Windows, paired-runtime, and provider-close lost-ack journeys remain explicit gaps.", "motivatingLinks": [ "https://github.com/stablyai/orca/pull/12355", "https://github.com/stablyai/orca/issues/13860", @@ -14099,6 +14393,8 @@ "invariant": "A settled Dispatch may close only its one coordinator-created terminal lease. Explicit reuse, real user input, retain, identity or host change, ambiguity, and another resource for the same exact host/pane/process must fence closure. Once the authoritative owning provider positively excludes the resource's exact immutable process incarnation, even an external, user-owned, or transferred dead resource must converge to released without any process close. Unknown host scope, missing incarnation metadata, or unavailable inventory must remain retained. Exact terminal-close persistence must settle when a host partition omits renderer-owned layout state. Output preservation and the requested-to-releasing transition are atomic, archives remain readable without the provider file, retries resume idempotently, and orchestration reset removes archive and authority state.", "oracle": "Record release intent for a settled owner, attempt exact reuse before close, and require worker-start to fail with terminal_release_in_progress while the terminal stays open; then release the original owner exactly once. Race retain and real user input against a controlled archive promise and require no committed archive or close. Rebase a closed web-terminal host partition without terminalLayoutsByTabId and require the persistence write to complete while preserving host-authoritative membership; replay a valid legacy retirement under the same omission and require exact membership removal plus revision advancement. For retained external, user-owned, transferred, stopped, and abandoned resources, run one fresh inventory against the exact local/WSL or SSH provider: an exact live incarnation and every unknown inventory shape stay retained, while positive absence atomically sets ownership_state and release_state to released with processAction none and zero closeTerminal calls. Change host or process identity and inject duplicate resource evidence to require retention. Freeze a structured transcript, delete its source file, and require archived worker-read to return the same bounded redacted messages. Restart a pending mutation, reset orchestration state, and create 50 resources while asserting replay convergence, zero orphan rows, two-query worker listing, and no unrelated close.", "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts src/main/runtime/orca-runtime-terminal-handle-incarnation.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts", + "ORCA_BACKGROUND_LAUNCH=1 mobile/node_modules/.bin/vitest run --config mobile/vitest.config.ts mobile/src/session/mobile-worker-takeover-send-sites.test.ts mobile/src/terminal/worker-terminal-takeover-report.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/pty-inventory-liveness-verdict.test.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot", "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot", "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot", @@ -14107,6 +14403,9 @@ "pnpm run build:cli && SKIP_BUILD=1 pnpm exec playwright test tests/e2e/orchestration-worker-settlement-release-cli.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1" ], "testFiles": [ + "src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts", + "mobile/src/session/mobile-worker-takeover-send-sites.test.ts", + "mobile/src/terminal/worker-terminal-takeover-report.test.ts", "src/main/runtime/pty-inventory-liveness-verdict.test.ts", "src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts", "src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts", @@ -14119,6 +14418,20 @@ "tests/e2e/orchestration-worker-settlement-release-cli.spec.ts" ], "assertionRefs": [ + { + "file": "src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts", + "assertions": [ + "a handle-addressed phone report fences %s worker release", + "mobile %s bytes do no orchestration database work" + ] + }, + { + "file": "mobile/src/session/mobile-worker-takeover-send-sites.test.ts", + "assertions": [ + "%s reports on its send target once per handle per 30 seconds", + "%s never reports takeover" + ] + }, { "file": "src/main/runtime/pty-inventory-liveness-verdict.test.ts", "assertions": [ @@ -18378,7 +18691,7 @@ "providers": ["ssh"], "coveredPlatforms": ["macos", "linux"], "coveredProviders": ["ssh"], - "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075. Deterministic remote Codex fixture validation passed three normal restores and three forced reconnects with zero retries on merged main plus the replay probe correction (run 34050117471). The original forced-reconnect probe missed nonempty replay returned in pty:spawn reattach replies. Routine coverage now includes both modes by default; real Codex service execution remains opt-in.", + "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075. Deterministic remote Codex fixture validation passed three normal restores and three forced reconnects with zero retries on merged main plus the replay probe correction (run 34050117471). The original forced-reconnect probe missed nonempty replay returned in pty:spawn reattach replies. Routine coverage now includes both modes by default; real Codex service execution remains opt-in. The added five-pane input spec passed in Linux CI run 34033353595, and diagnostic run 34034754815 reproduced real input loss during scrollback replay; application fix #19075 (98b0c329ff3) has since merged and this spec now guards it.", "motivatingLinks": [ "https://github.com/stablyai/orca/issues/18018", "https://github.com/stablyai/orca/pull/18546", @@ -18396,7 +18709,8 @@ "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1", "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10", "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-codex-display-artifacts-repro.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1", - "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts" + "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts", + "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1" ], "testFiles": [ "tests/e2e/ssh-docker-transport-drop-recovery.spec.ts", @@ -18408,7 +18722,8 @@ "tests/e2e/helpers/electron-process-shutdown.unit.test.ts", "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts", "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts", - "tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts" + "tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts", + "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts" ], "assertionRefs": [ { @@ -18460,6 +18775,12 @@ "five flooding SSH panes remain below unchanged 2500ms soft and 5000ms hard freeze budgets during bulk reopen and two double-animation-frame view changes" ] }, + { + "file": "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts", + "assertions": [ + "five distinct SSH PTYs acknowledge actual keyboard input after two rendered hide/reopen cycles while all five producers flood" + ] + }, { "file": "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts", "assertions": [ @@ -18508,7 +18829,7 @@ }, "flakeHistory": { "status": "flaky", - "evidence": "Baseline: ten enabled tests passed, two fixme skipped, worker teardown timed out (7.3m). After pipe cleanup: ten passed and worker exited cleanly (5.2m); two half-open repeats passed (1.7m). The formerly skipped thaw-input case failed before its recovered-authority wait and passed 1+3 executions afterward (56.9s + 2.6m). Flood failed both its original input oracle and a strengthened producer-completion oracle after recovery." + "evidence": "Baseline: ten enabled tests passed, two fixme skipped, worker teardown timed out (7.3m). After pipe cleanup: ten passed and worker exited cleanly (5.2m); two half-open repeats passed (1.7m). The formerly skipped thaw-input case failed before its recovered-authority wait and passed 1+3 executions afterward (56.9s + 2.6m). Flood failed both its original input oracle and a strengthened producer-completion oracle after recovery. Five-pane input diagnostics additionally failed 1/5 in run 34034754815: the intended focused PTY emitted the full input, the replay guard discarded 31 characters, and the remote ACK contained exactly the remaining suffix. PR #19075 addresses that application bug; passing repetitions alone do not establish its resolution." }, "redGreenEvidence": { "status": "partial", @@ -18524,6 +18845,7 @@ "Collect CI runtime and flake history plus product red/green evidence before blocking." ], "knownGaps": [ + "Five-pane simultaneous flood input was reproduced as a real application bug in run 34034754815 (replay discarded the first 31 characters of correctly focused keyboard input); fix #19075 (98b0c329ff3) merged and the spec now guards it, but the retries: 0 Docker SSH lane is the only repetition evidence against the merged fix so far. Freeze performance coverage was restored separately in #19081, with its isolated headless timer-lag outlier still documented.", "The disconnected 48MB flood still loses its relay channel: original post-flood input marker failed in 60s, and waiting for the finite producer completion marker failed in 120s. It remains an explicit #18018 fixme reproduction; frozen-host input is re-enabled after four successful runs.", "Linux headed CI covers the bulk-open freeze reproduction; Windows clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not covered by that result.", "Some legacy assertions inspect terminal serialization or backing state rather than rendered DOM; no blanket visual coverage claim.", diff --git a/config/scripts/idle-cpu-renderer-scale-fixture.mjs b/config/scripts/idle-cpu-renderer-scale-fixture.mjs index 4759b768e27..8f6490c2bd2 100644 --- a/config/scripts/idle-cpu-renderer-scale-fixture.mjs +++ b/config/scripts/idle-cpu-renderer-scale-fixture.mjs @@ -1,6 +1,6 @@ export async function configureRendererScaleFixture(page, options, repoPath) { return page.evaluate( - ({ agentsPerWorktree, lineageDepth, repoPath }) => { + ({ agentsPerWorktree, subagentsPerAgent, lineageDepth, repoPath }) => { const store = window.__store if (!store) { throw new Error('window.__store is not available') @@ -98,7 +98,18 @@ export async function configureRendererScaleFixture(page, options, repoPath) { { state: 'working', prompt: `Idle CPU agent ${worktreeIndex + 1}.${agentIndex + 1}`, - agentType + agentType, + ...(subagentsPerAgent > 0 + ? { + subagents: Array.from({ length: subagentsPerAgent }, (_, index) => ({ + id: `child-${index}`, + state: 'working', + startedAt: fixtureNow, + agentType, + description: `Subagent ${worktreeIndex + 1}.${agentIndex + 1}.${index + 1}` + })) + } + : {}) }, agentType, { updatedAt: fixtureNow, stateStartedAt: fixtureNow }, @@ -116,10 +127,16 @@ export async function configureRendererScaleFixture(page, options, repoPath) { expandedLineageGroups: lineageParentIds.size, agentsPerWorktree, seededAgentRows, + seededSubagentRows: seededAgentRows * subagentsPerAgent, orderedWorktreeIds: worktrees.map((worktree) => worktree.id) } }, - { agentsPerWorktree: options.agentsPerWorktree, lineageDepth: options.lineageDepth, repoPath } + { + agentsPerWorktree: options.agentsPerWorktree, + subagentsPerAgent: options.subagentsPerAgent ?? 0, + lineageDepth: options.lineageDepth, + repoPath + } ) } diff --git a/config/scripts/locale-collator-sort-benchmark.mjs b/config/scripts/locale-collator-sort-benchmark.mjs index 3d34466532a..1a0b08e1643 100644 --- a/config/scripts/locale-collator-sort-benchmark.mjs +++ b/config/scripts/locale-collator-sort-benchmark.mjs @@ -129,6 +129,7 @@ for (const count of [36, 50, 250]) { const issues = makeJiraIssues(count) const before = () => [...issues] + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Baseline measures per-comparison setup against a reused collator. .sort((a, b) => a.key.localeCompare(b.key, undefined, { numeric: true })) .map((issue) => issue.key) const after = () => sortJiraIssues(issues, 'key', 'asc').map((issue) => issue.key) @@ -142,6 +143,7 @@ for (const count of [36, 50, 250]) { for (const count of [10, 50, 250]) { const values = makeBaseSensitivityValues(count) const before = () => + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Baseline measures per-comparison setup against a reused collator. [...values].sort((a, b) => a.localeCompare(b, undefined, { sensitivity: 'base' })) const after = () => [...values].sort(compareBaseSensitivityLocaleText) assertSameOrder(before, after, `base ${count}`) diff --git a/config/scripts/native-chat-live-session-benchmark.ts b/config/scripts/native-chat-live-session-benchmark.ts index c21ae759a7f..5983dc2f4e7 100644 --- a/config/scripts/native-chat-live-session-benchmark.ts +++ b/config/scripts/native-chat-live-session-benchmark.ts @@ -117,7 +117,7 @@ function blockContent(message: NativeChatMessage): string { if (block.type === 'tool-result') { return block.output } - return block.path ?? block.url ?? block.alt ?? '' + return block.type === 'image-ref' ? (block.path ?? block.url ?? block.alt ?? '') : block.groupId } function messageWeight(message: NativeChatMessage, content: string): number { diff --git a/config/scripts/orchestration-skill-guidance.test.mjs b/config/scripts/orchestration-skill-guidance.test.mjs index ce501954322..c15e3e93ea8 100644 --- a/config/scripts/orchestration-skill-guidance.test.mjs +++ b/config/scripts/orchestration-skill-guidance.test.mjs @@ -168,9 +168,10 @@ describe('orchestration kernel', () => { expect(kernel).toContain( '`projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv' ) - expect(kernel).toContain( - 'An `inspect` `nextAction` on a `live` row with `attention.requiresAction` false is informational, not a command to re-run: keep waiting with `check --wait`' - ) + // Unverifiable workers can still owe release; the guide must explain the action itself. + expect(kernel).toContain('A `none` `nextAction` has no argv to run') + expect(kernel).toContain('read `liveness.reason` and keep waiting with `check --wait`') + expect(kernel).toContain('Absence never earns an argv; settlement and pending work still do') expect(kernel).toContain('choose `worker-stop` or `worker-abandon`') }) diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index 1c926b3622a..4f012b105b4 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -168,6 +168,9 @@ describe('PR E2E gate contract', () => { expect(changedRun.env.TEST_FILES_JSON).toBe('${{ inputs.test_files }}') expect(changedRun.run).toContain('. != "tests/e2e/ssh-startup-exec-readiness.spec.ts"') expect(changedRun.run).toContain('. != "tests/e2e/paired-startup-exec-readiness.spec.ts"') + expect(changedRun.run).toContain( + '. != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts"' + ) expect(changedRun.run).toContain('. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"') expect(changedRun.run).toContain('if [ "${#TEST_FILES[@]}" -eq 0 ]') expect(changedRun.run).toContain('grep -l \'@headful\' "${TEST_FILES[@]}"') diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs index b93a8e27411..fc0d628ab78 100644 --- a/config/scripts/run-ssh-docker-e2e.mjs +++ b/config/scripts/run-ssh-docker-e2e.mjs @@ -63,6 +63,7 @@ const result = spawnSync( 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', 'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts', + 'tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts', 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts', 'tests/e2e/ssh-docker-half-open-link.spec.ts', 'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts', diff --git a/config/ts-nocheck-baseline.txt b/config/ts-nocheck-baseline.txt index b770b06f827..e897af7387c 100644 --- a/config/ts-nocheck-baseline.txt +++ b/config/ts-nocheck-baseline.txt @@ -34,7 +34,7 @@ src/main/runtime/orca-runtime-create-terminal-side-effect-command-code-detector. src/main/runtime/orca-runtime-create-terminal.ts src/main/runtime/orca-runtime-deliver-pending-messages.ts src/main/runtime/orca-runtime-emit-daemon-pty-transient-fact.ts -src/main/runtime/orca-runtime-fence-automation-owner.ts +src/main/runtime/orca-runtime-automation-operations.ts src/main/runtime/orca-runtime-file-commands.ts src/main/runtime/orca-runtime-fit-override-listeners.ts src/main/runtime/orca-runtime-focus-terminal.ts diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index ebee3673b77..724be685ea7 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 42m + + downloads: 44m @@ -15,7 +15,7 @@ downloads downloads - 42m - 42m + 44m + 44m diff --git a/docs/reference/renderer-agent-status-performance.md b/docs/reference/renderer-agent-status-performance.md index 8ed43d868ce..cffad695d25 100644 --- a/docs/reference/renderer-agent-status-performance.md +++ b/docs/reference/renderer-agent-status-performance.md @@ -87,13 +87,25 @@ bundled prototype, the fixture without seeded agents fell from 8,518 listeners to 1,218; with 100 visible agent rows the candidate mounted 1,618. Compare against the census in "Baseline on `main`", which the harness reports directly. -### Share working-spinner phase without per-element animation queries +### Share working-spinner phase without synchronous mount queries Working rows keep the existing compositor-driven CSS animation and shared -visual phase. Each mount derives one negative animation delay from the document -timeline instead of querying `getAnimations()` and mutating the animation start -time. This removes per-row Web Animations setup from dense status transitions -without adding a JavaScript animation clock. +visual phase. `animationstart` anchors each animation to document time zero. +Deferring the animation query until that event avoids a synchronous style flush +at each mount and restores the shared phase after `display:none` or a motion +preference change. A negative mount-time delay cannot preserve that phase after +an animation restarts. + +### Bound spinner animation overhead + +Working rings keep compositor-driven CSS animation, but repeat the animation +once per day rather than once per second. The same 12 steps per second now +avoid recurring React animation-iteration dispatch. The existing stationary +wrapper and ring rendering stay unchanged. Offscreen containment was evaluated +and rejected after a pixel regression at low zoom on 1x displays. + +The history, isolated measurements, full-app workspace/agent/subagent benchmark, +and limitations are documented in [Spinner rendering performance](./spinner-rendering-performance.md). ### Fold a burst in event order diff --git a/docs/reference/spinner-rendering-performance.md b/docs/reference/spinner-rendering-performance.md new file mode 100644 index 00000000000..4340027474b --- /dev/null +++ b/docs/reference/spinner-rendering-performance.md @@ -0,0 +1,203 @@ +# Spinner rendering performance + +## ELI5 + +Imagine a wheel that tells the front desk every time it completes a lap. The +front desk is also handling your typing. CSS already turns the wheel for us, +but React still receives its once-per-second lap notifications. + +We put a day's worth of laps into one animation. The wheel moves at the same +speed, while sending one lap notification a day. Drawing visible wheels still +costs something. This removes recurring bookkeeping from the input thread; it +does not make rendering or the rest of Orca free. + +## How this builds on earlier changes + +| Change | What it achieved | Remaining cost | +| ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- | +| [#9380](https://github.com/stablyai/orca/pull/9380): shared JavaScript clock | Reduced frame-pipeline CPU in the original one-agent measurement | Wrote each spinner's style 12 times per second on the input thread | +| [#12359](https://github.com/stablyai/orca/pull/12359): compositor CSS rotation | Removed those recurring JavaScript style writes; fixed the reported typing regression | React still receives CSS iteration events | +| [#13987](https://github.com/stablyai/orca/pull/13987): synchronize on animationstart | Avoided a synchronous style query at every mount | Steady-state animation overhead stayed the same | +| This change | Preserves both later fixes and removes almost all iteration boundaries | Compositing, other app work, mount/reveal work, and a daily iteration boundary remain | + +The historical measurements in #12359 reported 41 rings causing about 490 style +writes per second, with typing input-delay p90 of 363 ms versus 19 ms when those +writes stopped. Those are historical production measurements, not numbers from +this benchmark or a direct comparison with today's app. + +## Implementation + +The production change is entirely in CSS. `AgentWorkingSpinner`, its callers, +markup, border, animation-start handler, and reduced-motion behavior stay the +same. No DOM node, pseudo-element, containment boundary, timer, observer, or +JavaScript animation loop is added. + +The transform travels 86,400 turns in 86,400 seconds with 1,036,800 steps: exactly +one revolution and 12 steps per second. `animationstart` sets `startTime = 0` as +before, preserving shared phase after mount and animation restart. The step +count is a timing-function parameter, not a million-entry keyframe list. + +React installs delegated `animationiteration` listeners even when the component +has no iteration handler. A native 2.2-second trace of 200 isolated rings counted +400 iteration events and 800 JavaScript calls before the change, versus zero of +either with the long cycle. That trace installed no animation-event listener. +These are event dispatches, not component rerenders or 400 separate OS wakeups. + +## Full-app benchmark + +The opt-in Playwright benchmark launches a fresh, hidden Orca app for each +scenario. It creates real Git workspaces and seeds working statuses through the +existing renderer fixture, including in-process subagent data. It renders the +normal sidebar, virtualizer, lineage, agent rows, tabs, and terminal. + +| Scenario | Git workspaces | Root agents | Subagents | Mounted / visible rings | Layout | +| ------------- | -------------: | ----------: | --------: | ----------------------: | -------------------------------------------- | +| `one-agent` | 1 | 1 | 0 | 3 / 3 | One working agent | +| `one-family` | 1 | 2 | 4 | 8 / 8 | All family rows expanded | +| `200-flat` | 200 | 400 | 800 | 162 / 15 | Normal virtualization; 23 workspaces mounted | +| `200-lineage` | 200 | 400 | 800 | 1,401 / 15 | Expanded lineage; all 200 workspaces mounted | + +Measurement-only styles switch between the original one-second cycle and the +new long cycle on the same elements. The real React root, callers, status data, +and app stay the same. The reported run alternates A/B and B/A, with four +ten-second CPU samples per variant after warmup. CPU samples use cumulative +Electron process CPU and CDP main-thread task/script/style/layout metrics. No +renderer polling, screenshots, or benchmark iteration listeners run during +those CPU windows. No samples are discarded. + +Typing is measured separately using the existing paced terminal-typing probe: +64 keys at 113 ms cadence, twice per variant, after two seconds of warmup with +status traffic. Status updates arrive in groups of up to eight every 200 ms. +Keys pass through the DOM, real PTY, and xterm. A sidecar timestamps arrival at +the PTY, and a bounded terminal-buffer scan observes each echo. Missing input +or echoes fail the benchmark. Echo measurements include the 10 ms scan interval; +they do not measure native display presentation. Native animation traces also +run separately from CPU and typing samples. + +The statuses are deterministic test data, not hundreds of paid model sessions. +The test exercises UI cost under agent-status traffic, not the compute or network +cost of model inference, SSH traffic, or hundreds of streaming PTYs. + +## Results + +CPU values are medians of four samples. "CPU ms/s" means milliseconds of +processor time used in one wall-clock second: 100 ms/s is about 10% of one CPU +core. Renderer + GPU-process CPU includes their other app work and CPU used by +the graphics process; it is not GPU hardware utilization or whole-machine CPU. +The main thread handles input and is included in renderer CPU, not extra work. +Echo p90 means 90% of sampled keys were observed within that time; ranges show +the two runs, not confidence intervals. No keys or echoes were missing. + +| Scenario | Renderer + GPU CPU ms/s, old → new | Main-thread ms/s, old → new | Echo p90 ms, old → new | +| ------------- | ---------------------------------: | --------------------------: | ---------------------- | +| `one-agent` | 37.0 → 38.2 | 5.4 → 3.5 | 19 → 18–19 | +| `one-family` | 46.2 → 44.6 | 7.8 → 4.4 | 17–19 → 18–19 | +| `200-flat` | 141.6 → 122.8 | 28.2 → 16.3 | 26–28 → 26–28 | +| `200-lineage` | 324.6 → 295.6 | 140.8 → 70.0 | 159–239 → 93–160 | + +The consistent gain is less main-thread work: about 35%, 43%, 42%, and 50% +less in these four scenarios. Native 2.2-second traces counted 6, 16, 324, and +2,802 iteration events before, and zero in each new variant, without adding an +iteration listener. That avoided work also exists in Orca itself, independently +of the isolated fixture and CPU noise. + +Total CPU was roughly unchanged in the one-worktree cases. In this run it fell +13% with normal virtualization and 9% with expanded lineage; seven of eight +paired large-case CPU samples favored the change. These percentages are not +universal: a shorter three-variant ablation measured flat-list CPU at 89.0 ms/s before and +108.0 ms/s with the long cycle, while main-thread time still fell from 26.3 to +17.4 ms/s. The repeatable main-thread reduction is stronger evidence than a +single total-CPU percentage. + +Typing was similar in the small and flat-list cases. Expanded-lineage echo p90 +improved in the final run, but a shorter ablation had similar before/after +latencies. No general typing speedup or statistical non-regression guarantee +is established by these short experiments. + +### All CPU samples + +Values are rounded to one decimal and listed by round, with no outliers removed. +The first new small-case samples were higher than their paired baselines; they +remain included. CPU and typing were sampled separately. + +| Scenario | Version | Renderer + GPU CPU ms/s | Main-thread ms/s | +| ------------- | ------- | -------------------------- | -------------------------- | +| `one-agent` | Old | 37.8, 36.2, 26.2, 39.6 | 6.5, 5.2, 4.8, 5.7 | +| `one-agent` | New | 53.3, 35.8, 37.5, 39.0 | 6.7, 2.8, 3.0, 4.1 | +| `one-family` | Old | 46.3, 46.1, 47.9, 44.6 | 7.8, 7.6, 9.8, 7.7 | +| `one-family` | New | 53.8, 45.4, 42.5, 43.8 | 6.8, 4.5, 3.1, 4.3 | +| `200-flat` | Old | 142.4, 140.8, 147.0, 136.5 | 28.3, 28.1, 32.2, 27.0 | +| `200-flat` | New | 122.9, 97.2, 122.7, 126.7 | 19.2, 10.7, 16.6, 16.0 | +| `200-lineage` | Old | 317.9, 385.2, 315.9, 331.2 | 134.4, 159.6, 133.9, 147.3 | +| `200-lineage` | New | 318.6, 256.9, 296.5, 294.7 | 89.2, 60.8, 71.6, 68.3 | + +## Reproduce + +```sh +ORCA_BACKGROUND_LAUNCH=1 pnpm bench:spinners --sample-ms=5000 +ORCA_BACKGROUND_LAUNCH=1 pnpm bench:spinners --verify-only --scale-factor=1 +ORCA_BACKGROUND_LAUNCH=1 pnpm bench:spinners --verify-only --scale-factor=2 +ORCA_BACKGROUND_LAUNCH=1 ORCA_SPINNER_BENCH=1 ORCA_SPINNER_KEYS=64 \ + pnpm test:e2e spinner-workspace-perf.spec.ts --workers=1 +``` + +The full-app command rebuilds in `e2e` mode. For a fresh build already made with +`pnpm exec electron-vite build --mode e2e`, `SKIP_BUILD=1` reuses it. Do not reuse +an old launch-policy build. `ORCA_SPINNER_SAMPLE_MS`, `ORCA_SPINNER_ROUNDS`, +`ORCA_SPINNER_KEYS`, `ORCA_SPINNER_KEY_CADENCE_MS`, `ORCA_SPINNER_VARIANTS`, and +`ORCA_SPINNER_OUTPUT` control the experiment. `ORCA_SPINNER_CPU=0` repeats only +typing; `--grep one-agent` selects one scenario. Reports, native traces, typing +sidecars, and CDP screenshots are written under `.bench-fixtures/`. Run one +benchmark at a time, without concurrent builds or tests. + +The optional `contained` variant retains the rejected offscreen experiment for +ablation. It adds `content-visibility:auto` to the existing wrapper through +measurement-only styles. It is not enabled in production or the default +benchmark comparison. + +## Visual and behavioral checks + +Both 1x and 2x display-density checks passed 720 ring comparisons each: 6/8 px +rings, light/dark themes, supported zoom extremes, all 12 phases, long elapsed +times, and the daily wrap. The comparison pauses each animation and sets its +`currentTime`, so the long-elapsed and daily-wrap cases exercise the deterministic +style path rather than a running compositor animation. Against that path the +tolerance is one channel level for floating-point antialias rounding. A running +animation at multi-hour ages can differ by a few channels on the ring edge — a +fraction-of-a-pixel antialias difference at large accumulated angles, not a phase +or shape change. Checks also cover shared phase, reduced motion, initial offscreen +reveal, repeated scroll-away/reveal, and `display:none` restoration. + +## Limits and rejected approaches + +Adding `content-visibility:auto` to the existing stationary wrapper saved more +CPU at large mounted counts, but a 1x display check found a one-pixel shift at +the minimum UI zoom. That containment change is excluded. A previous +pseudo-element version also regressed typing latency in the virtualized list. +Neither prototype's CPU or typing numbers describe the final patch. + +An initial typing run used a 100 ms key cadence, which can repeatedly align with +200 ms status bursts. Follow-up runs use 113 ms, more keys, and two seconds of +warmup under status traffic. This reduces timing bias; it does not excuse a +regression. CPU measurements run separately and do not depend on key cadence. + +An early isolated test suggested a 31% process-CPU reduction that a longer audit +did not reproduce. The longer isolated audit measured original 104.04 versus +long-cycle 92.32 CPU ms/s, and main-thread 10.08 versus 0.24 ms/s. A fixture with +every ring far offscreen and containment enabled could also approach idle; that +is not representative of Orca with visible animations. Neither result justifies +claiming "free spinners" or a universal CPU percentage. Virtualized, unmounted +rows already cost nothing, and this patch does not add offscreen culling. + +All local measurements use an Apple M4 (10 cores), macOS, Electron 43.4.1 / +Chromium 150.0.7871.224. Native windows stay hidden and unfocused; +benchmark-only settings disable background throttling to exercise the frame +pipeline. These are not visible-window power measurements. No battery benefit +is established. Linux/Windows need their own runtime measurements. The +renderer-only change does not alter SSH execution, wire data, status semantics, +Git operations, or folder-workspace ownership. + +Animated PNGs, masks, layer promotion, CSS sprites, individual `rotate`, and +containment on the rotating element were also explored. Shared images added +raster work and regressed the single-ring case; sprites reintroduced per-frame +style work. They did not meet the appearance and responsiveness requirements. diff --git a/docs/review-evidence/pr-19217/README.md b/docs/review-evidence/pr-19217/README.md new file mode 100644 index 00000000000..ca664ae0c9a --- /dev/null +++ b/docs/review-evidence/pr-19217/README.md @@ -0,0 +1,46 @@ +# Structured worktree status validation + +Validated on September 7, 2026 in a background Electron dev instance of +`pr19217-review-r2`, based on `ce1024096b` with the source-adapter refactor. +CDP app identity confirmed the checkout; screenshots show the full hidden renderer. +The command output is the real `orca worktree ps --json` response reduced to status, +agent state, provider, and pane key for readability. + +## Functional correctness + +A real Codex structured session appeared as `working` in `worktree.ps` while the +sidebar showed working. Closing its chat tab removed that exact session's row and +returned the worktree to `active`. A different completed chat remained present, +confirming that closure removed only the selected session. + +- [Working: CLI and sidebar](working.png) +- [Closed: CLI and sidebar](closed.png) + +The disappearing session is `codex_40677067_f492_4d7d_86dd_ec566ede04c3`. +The host's held-session roster controls eligibility; its retained broadcast cache +is history, not a roster. Failed eviction intentionally keeps an entry for retry. + +## Architecture + +PTY reconciliation and process admission belong to the PTY source adapter. +Structured input comes from the current host's held-session projections. One +admitted collection feeds row shaping and worktree aggregation, with no structured +boolean bypass. PTY hooks and retained reports still arrive independently, so their +precedence and conservative remote evidence rules remain necessary. No second +persistent status store or provider polling was introduced. + +## Validation and limits + +Independent final review found no proven issues. Runtime, host lifecycle, status +feed and source-admission suites passed: 1,344 tests, one skipped. Node typecheck, +targeted lint and diff checks passed. Ablating the runtime call to enumerate +retained history caused the executable call-site test to fail with two rows where +one was expected; restoring the live accessor passed both call-site tests. + +Live screenshots prove Codex working and closure on macOS. Claude provider turns, +approval/input states, live Windows/Linux/WSL/SSH/relay/mobile scenarios and +release-scale latency/heap measurements remain unverified. Existing tests cover +remote/WSL evidence, monitoring precedence and lifecycle cases. The existing +30-minute freshness rule and CLI activity timestamps are preserved; complete +CLI/sidebar timing parity is not claimed. The wire keeps its existing row shape +and status vocabulary; mobile receives the new rows without a new opcode. diff --git a/docs/review-evidence/pr-19217/closed.png b/docs/review-evidence/pr-19217/closed.png new file mode 100644 index 00000000000..7f1d8160844 Binary files /dev/null and b/docs/review-evidence/pr-19217/closed.png differ diff --git a/docs/review-evidence/pr-19217/working.png b/docs/review-evidence/pr-19217/working.png new file mode 100644 index 00000000000..8ee57052134 Binary files /dev/null and b/docs/review-evidence/pr-19217/working.png differ diff --git a/mobile/src/session/mobile-native-chat-permission-send.test.ts b/mobile/src/session/mobile-native-chat-permission-send.test.ts index f74289d97ab..1525f090029 100644 --- a/mobile/src/session/mobile-native-chat-permission-send.test.ts +++ b/mobile/src/session/mobile-native-chat-permission-send.test.ts @@ -1,3 +1,8 @@ +// Takeover RPCs have their own send-site integration tests; these fixtures script PTY acknowledgements. +vi.mock('../terminal/worker-terminal-takeover-report', () => ({ + reportWorkerTerminalUserInput: vi.fn() +})) + import { createElement } from 'react' import { act, create, type ReactTestRenderer } from 'react-test-renderer' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' diff --git a/mobile/src/session/mobile-native-chat-send.test.ts b/mobile/src/session/mobile-native-chat-send.test.ts index a3b3c1e720e..c13841f7f77 100644 --- a/mobile/src/session/mobile-native-chat-send.test.ts +++ b/mobile/src/session/mobile-native-chat-send.test.ts @@ -309,10 +309,13 @@ describe('typeMobileNativeChatCommandWithOutcome', () => { await expect(result).resolves.toBe('accepted') expect( - vi.mocked(client.sendRequest).mock.calls.map((call) => { - const params = call[1] as { text: string; enter: boolean } - return { text: params.text, enter: params.enter } - }) + vi + .mocked(client.sendRequest) + .mock.calls.filter(([method]) => method === 'terminal.send') + .map((call) => { + const params = call[1] as { text: string; enter: boolean } + return { text: params.text, enter: params.enter } + }) ).toEqual( ['\x15', '/', 'm', 'o', 'd', 'e', 'l', '\r'].map((text) => ({ text, @@ -338,7 +341,12 @@ describe('typeMobileNativeChatCommandWithOutcome', () => { await vi.runAllTimersAsync() await result - const params = vi.mocked(client.sendRequest).mock.calls.map((call) => call[1]) as Array<{ + // Why the filter: an accepted send also fires the unawaited takeover report, which is not a + // terminal.send and carries no draft. + const params = vi + .mocked(client.sendRequest) + .mock.calls.filter((call) => call[0] === 'terminal.send') + .map((call) => call[1]) as Array<{ text: string resolvedLaunchDraft?: { text: string; createdAt: number } }> diff --git a/mobile/src/session/mobile-native-chat-send.ts b/mobile/src/session/mobile-native-chat-send.ts index 16f4aac2d3e..22c44f3eb84 100644 --- a/mobile/src/session/mobile-native-chat-send.ts +++ b/mobile/src/session/mobile-native-chat-send.ts @@ -1,3 +1,4 @@ +import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report' import type { RpcClient } from '../transport/rpc-client' import { isRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' import { isLogicalClientCutoverError } from '../transport/stable-logical-rpc-client' @@ -64,7 +65,11 @@ export async function sendMobileNativeChatMessageWithOutcome( // pins the composer for twice as long. { timeoutMs, budgetSpansConnect: true } ) - return isTerminalSendRpcAccepted(response) ? 'accepted' : 'rejected' + if (!isTerminalSendRpcAccepted(response)) { + return 'rejected' + } + reportWorkerTerminalUserInput(args.client, args.terminal) + return 'accepted' } catch (error) { // Why: a logical relay↔direct cutover rejects the in-flight send without // knowing whether its frame reached the wire (the desktop may have delivered diff --git a/mobile/src/session/mobile-session-route-parity.test.ts b/mobile/src/session/mobile-session-route-parity.test.ts index bc951bfa206..3a6b04661de 100644 --- a/mobile/src/session/mobile-session-route-parity.test.ts +++ b/mobile/src/session/mobile-session-route-parity.test.ts @@ -66,11 +66,11 @@ const HEAD_MAIN_HOOK_SHA256 = '10071240ef9edafc2b9c8bed73be83dceaf7828e3b29f17da const HEAD_HOOK_BINDING_SHA256 = '1dadb8c3dc0573ea20659ce7251629669e618dd0effaeac3a4536b29c2e865a1' const HEAD_CALLBACK_IDENTITY_SHA256 = '2a9e4825df007f6ef53b81aa5004991d6318eee7507b44d625c07e630be432eb' -const HEAD_CALLBACK_BODY_SHA256 = '22103ba85a86e3a3fcb80a7509c7a455d79863010cde3af02db6565b55e3ebe9' +const HEAD_CALLBACK_BODY_SHA256 = 'af7f3c62954250d4be7ee432ecd10dc2689792aad8230fed2d1d68bbc892d776' const HEAD_EFFECT_SHA256 = 'd9ebfaabc1e79773cdada7ab370b20459ed972f1f8edce1652199f4d0391cd13' const HEAD_CONTENT_HOOK_SHA256 = '9c3b612fef3f370d66873aefdbe1d701f20cb64ded31fef5cc45fde6f8189581' const HEAD_NESTED_FUNCTION_SHA256 = - '536c72b233c813bb0cea164b090bdce5406ceb965bbc5b83c1f89b89b46f3821' + 'fde6679349ab2b8c30c7e627841ff99bd1dd24441ee95323d0aa70230422ae24' const HEAD_NATIVE_REGISTRATION_SHA256 = 'cab85e4e4a3f43289ba93ddea9ccce57aea83e0bf14fd1620a965aad0c1cb49e' const HEAD_NATIVE_REMOVAL_SHA256 = diff --git a/mobile/src/session/mobile-worker-takeover-send-sites.test.ts b/mobile/src/session/mobile-worker-takeover-send-sites.test.ts new file mode 100644 index 00000000000..2700d8d24fe --- /dev/null +++ b/mobile/src/session/mobile-worker-takeover-send-sites.test.ts @@ -0,0 +1,267 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { beforeEach, afterEach, expect, it, vi } from 'vitest' +import type { RpcClient } from '../transport/rpc-client' +import { resetWorkerTerminalTakeoverReportsForTest } from '../terminal/worker-terminal-takeover-report' +import { useMobileSessionTerminalSendActions } from './use-mobile-session-terminal-send-actions' +import { useMobileSessionTerminalInput } from './use-mobile-session-terminal-input' +import { useMobileTerminalPaste } from './use-mobile-terminal-paste' +import { useTerminalLiveInputCommit } from '../terminal/use-terminal-live-input-commit' +import { routeDictationTranscript } from '../terminal/terminal-live-dictation-routing' +import { + sendMobileNativeChatMessageWithOutcome, + clearMobileNativeChatInput +} from './mobile-native-chat-send' +import { sendMobileTerminalQueryReply } from '../terminal/mobile-terminal-query-reply' +import { createTerminalAndSendPrompt } from './pr-ai-triage-launch' +import { useMobileDiffReviewSendActions } from './use-mobile-diff-review-send-actions' +import { pasteMobileNativeChatImagePaths } from './mobile-native-chat-image-send' + +vi.mock('react-native', () => ({ Keyboard: { dismiss: vi.fn() } })) +vi.mock('../platform/haptics', () => ({ triggerError: vi.fn(), triggerSuccess: vi.fn() })) +vi.mock('expo-clipboard', () => ({ getStringAsync: async () => 'pasted text' })) +vi.mock('expo-file-system', () => ({ File: class {}, Paths: { cache: '/tmp' } })) +vi.mock('expo-image-manipulator', () => ({ ImageManipulator: {}, SaveFormat: {} })) + +const REPORT = 'orchestration.workerTerminalUserInput' +const ref = (current: T) => ({ current }) +const renderers: ReactTestRenderer[] = [] +function clientFixture() { + return { + sendRequest: vi.fn(async (method: string) => ({ + id: 'rpc', + ok: true as const, + result: + method === 'session.tabs.createTerminal' + ? { tab: { type: 'terminal', id: 'tab', terminal: 'term-1', title: 'test' } } + : method === REPORT + ? { changed: 1 } + : { send: { accepted: true } } + })) + } +} + +function mountSendSites(client: ReturnType, handle = 'term-1') { + const activeHandleRef = ref(handle) + const activeSessionTabTypeRef = ref('terminal') + const sendLiveTerminalInputRef = ref(async (_handle: string, _text: string) => false) + const scope = { + client, + clientRef: ref(client), + activeHandle: handle, + activeHandleRef, + activeSessionTabTypeRef, + connState: 'connected', + connStateRef: ref('connected'), + activeSessionTab: { type: 'terminal', terminal: handle }, + sendingRef: ref(false), + canSend: true, + deviceTokenRef: ref('phone'), + liveInputRef: ref(null), + commandInputRef: ref(null), + liveInputFocusTimerRef: ref(null), + sendLiveTerminalInputRef, + getSendCompletionGeneration: () => 0, + showToast: vi.fn(), + ptyModesRef: ref(new Map([[handle, { altScreen: true }]])), + terminalGestureInputBucketsRef: ref(new Map()), + terminalGestureInputQueuesRef: ref(new Map()), + terminalGestureInputInFlightRef: ref(new Set()), + bufferedTerminalDraftState: { + input: 'command', + beginBufferedTerminalDraftSend: vi.fn(), + restoreRejectedDraft: vi.fn(), + settleBufferedTerminalDraftSend: () => true + } + } + let actions!: ReturnType + let live!: ReturnType + let gestures!: ReturnType + let paste!: ReturnType + let diff!: ReturnType + function Harness() { + live = useTerminalLiveInputCommit({ + activeHandle: handle, + activeHandleRef, + activeSessionTabType: 'terminal', + activeSessionTabTypeRef, + connected: true, + liveInputRef: ref(null), + liveInputTerminalHandles: new Set([handle]), + liveInputTerminalHandlesRef: ref(new Set([handle])), + sendLiveTerminalInputRef, + setLiveInputCapture: vi.fn() + }) + actions = useMobileSessionTerminalSendActions({ + ...scope, + handleLiveInputAccessoryBytes: live.handleLiveInputAccessoryBytes + } as never) + gestures = useMobileSessionTerminalInput(scope as never) + paste = useMobileTerminalPaste({ + ...scope, + flushPendingLiveInputBeforeExternalSend: live.flushPendingLiveInputBeforeExternalSend, + getActiveWorktreeConnectionId: async () => null, + onError: vi.fn(), + onSuccess: vi.fn(), + refreshCanPaste: vi.fn() + } as never) + diff = useMobileDiffReviewSendActions({ + client: client as unknown as RpcClient, + connState: 'connected', + worktreeId: 'workspace', + screenState: { kind: 'loading' }, + setActionError: vi.fn(), + setSendSheet: vi.fn(), + saveCommentsAndReviewState: vi.fn() + } as never) + return null + } + act(() => { + renderers.push(create(createElement(Harness))) + }) + let text = '' + return { + 'live field': async () => { + text += 'x' + live.handleLiveInputChange({ nativeEvent: { text, isComposing: false } }) + await live.flushPendingLiveInputBeforeExternalSend(handle) + }, + 'live submit': () => live.handleLiveInputSubmit(), + 'live accessory': async () => { + live.handleLiveInputChange({ nativeEvent: { text: 'composing', isComposing: true } }) + await live.handleLiveInputAccessoryBytes({ bytes: '\x1b[A' }) + }, + 'raw accessory': () => actions.handleAccessoryKey({ bytes: '\x1b[A' } as never), + 'buffered submit': () => actions.handleSend(), + 'gesture arrows': async () => { + await gestures.handleTerminalInput(handle, '\x1b[A') + await gestures.flushTerminalGestureInput(handle) + }, + paste: () => paste(), + dictation: async () => { + const route = routeDictationTranscript('dictated text', true) + expect(route.kind).toBe('live-insert') + await actions.sendLiveTerminalInput(handle, route.text) + }, + 'native chat': () => + sendMobileNativeChatMessageWithOutcome({ + client: client as unknown as RpcClient, + terminal: handle, + text: 'hello' + }), + 'query reply': () => + sendMobileTerminalQueryReply({ + bytes: '\x1b[0n', + client, + clientId: 'phone', + connected: true, + handle, + hostSupportsQueryReplyInput: true, + subscribedTerminals: new Set([handle]) + }), + 'image heal': () => + clearMobileNativeChatInput({ + client: client as unknown as RpcClient, + terminal: handle, + clearInput: '\x15' + }), + 'image attachment': () => + pasteMobileNativeChatImagePaths({ + client, + terminal: handle, + deviceToken: 'phone', + imagePaths: ['/tmp/picture.png'], + followedByText: true + }), + 'PR triage': () => createTerminalAndSendPrompt(client, 'workspace', 'fix checks'), + 'diff review': () => diff.sendPromptToTerminal(handle, []), + programmatic: () => client.sendRequest('terminal.send') + } +} + +beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(1_000) + resetWorkerTerminalTakeoverReportsForTest() +}) +afterEach(() => { + act(() => { + for (const renderer of renderers.splice(0)) { + renderer.unmount() + } + }) + vi.useRealTimers() +}) + +const realSites = [ + 'live field', + 'live submit', + 'live accessory', + 'raw accessory', + 'buffered submit', + 'gesture arrows', + 'paste', + 'dictation', + 'native chat' +] as const +it.each(realSites)('%s reports on its send target once per handle per 30 seconds', async (site) => { + const client = clientFixture() + const sites = mountSendSites(client) + const invoke = async () => { + await act(async () => { + await sites[site]() + }) + } + const reports = () => client.sendRequest.mock.calls.filter(([method]) => method === REPORT) + await invoke() + await invoke() + expect( + client.sendRequest.mock.calls.filter(([method]) => method === 'terminal.send').length + ).toBeGreaterThanOrEqual(2) + expect(reports()).toHaveLength(1) + expect(reports()[0]).toEqual([REPORT, { terminal: 'term-1' }, expect.any(Object)]) + await vi.advanceTimersByTimeAsync(29_999) + await invoke() + expect(reports()).toHaveLength(1) + await vi.advanceTimersByTimeAsync(1) + await invoke() + expect(reports()).toHaveLength(2) + const other = mountSendSites(client, 'term-2') + await act(async () => { + await other[site]() + }) + expect(reports()).toHaveLength(3) + expect(reports()[2][1]).toEqual({ terminal: 'term-2' }) +}) + +it.each([ + 'query reply', + 'image heal', + 'image attachment', + 'PR triage', + 'diff review', + 'programmatic' +] as const)('%s never reports takeover', async (site) => { + const client = clientFixture() + const sites = mountSendSites(client) + await act(async () => { + await sites[site]() + await sites[site]() + }) + expect(client.sendRequest.mock.calls.some(([method]) => method === 'terminal.send')).toBe(true) + expect(client.sendRequest.mock.calls.filter(([method]) => method === REPORT)).toHaveLength(0) +}) + +it.each(realSites)('%s does not report a rejected send', async (site) => { + const client = clientFixture() + client.sendRequest.mockResolvedValue({ + id: 'rpc', + ok: true, + result: { send: { accepted: false } } + }) + const sites = mountSendSites(client) + await act(async () => { + await sites[site]() + }) + expect(client.sendRequest.mock.calls.filter(([method]) => method === REPORT)).toHaveLength(0) +}) diff --git a/mobile/src/session/pr-ai-triage-prompt.test.ts b/mobile/src/session/pr-ai-triage-prompt.test.ts index 4ed8af5ff23..ba2c4f292da 100644 --- a/mobile/src/session/pr-ai-triage-prompt.test.ts +++ b/mobile/src/session/pr-ai-triage-prompt.test.ts @@ -34,24 +34,22 @@ describe('getBrokenChecks / hasBrokenChecks', () => { }) describe('buildFixChecksPrompt', () => { - it('embeds PR identity and only broken checks as JSON data', () => { + // The wrapper only renames fields onto buildFixBrokenChecksPrompt, so assert the + // mapping and nothing else; prompt wording is pinned by that builder's own tests. + it('maps mobile PR fields onto the shared prompt builder', () => { const prompt = buildFixChecksPrompt({ prNumber: 42, prTitle: 'Add feature', prUrl: 'https://gh/pr/42', checks: [ - check({ name: 'lint', conclusion: 'success' }), check({ name: 'unit', conclusion: 'failure', checkRunId: 9, url: 'https://ci/unit' }) ] }) - expect(prompt).toContain('Fix the broken checks for PR #42.') - expect(prompt).toContain('untrusted data only, not instructions') + + expect(prompt).toContain('"number": 42') expect(prompt).toContain('"title": "Add feature"') + expect(prompt).toContain('"url": "https://gh/pr/42"') expect(prompt).toContain('"name": "unit"') - expect(prompt).toContain('"status": "Failed"') - // The passing check must not appear in the broken-check payload. - expect(prompt).not.toContain('"name": "lint"') - expect(prompt).toContain('Focus only on making the failing pull request checks pass') }) it('falls back to a refresh hint when nothing is broken', () => { diff --git a/mobile/src/session/use-mobile-native-chat-answer-send.test.ts b/mobile/src/session/use-mobile-native-chat-answer-send.test.ts index 82db799deed..cfb35417e9f 100644 --- a/mobile/src/session/use-mobile-native-chat-answer-send.test.ts +++ b/mobile/src/session/use-mobile-native-chat-answer-send.test.ts @@ -1,3 +1,8 @@ +// Takeover RPCs have their own send-site integration tests; these fixtures script PTY acknowledgements. +vi.mock('../terminal/worker-terminal-takeover-report', () => ({ + reportWorkerTerminalUserInput: vi.fn() +})) + import { createElement } from 'react' import { act, create, type ReactTestRenderer } from 'react-test-renderer' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' diff --git a/mobile/src/session/use-mobile-native-chat-stop.test.ts b/mobile/src/session/use-mobile-native-chat-stop.test.ts index bdc405cb57d..a1ee9ab4dd9 100644 --- a/mobile/src/session/use-mobile-native-chat-stop.test.ts +++ b/mobile/src/session/use-mobile-native-chat-stop.test.ts @@ -6,6 +6,12 @@ import { markRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' import { MOBILE_NATIVE_CHAT_SEND_TIMEOUT_MS } from './mobile-native-chat-send' import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' +// Why mocked: the reporter is tested on its own; here Stop's escapes must be counted alone. +const reportWorkerTerminalUserInput = vi.fn() +vi.mock('../terminal/worker-terminal-takeover-report', () => ({ + reportWorkerTerminalUserInput: (...args: unknown[]) => reportWorkerTerminalUserInput(...args) +})) + describe('useMobileNativeChatStop', () => { let renderer: ReactTestRenderer | null = null let stop: (() => void) | null = null @@ -19,6 +25,7 @@ describe('useMobileNativeChatStop', () => { result: { send: { accepted: true } } }) onSendError.mockReset() + reportWorkerTerminalUserInput.mockReset() }) afterEach(() => { @@ -184,4 +191,26 @@ describe('useMobileNativeChatStop', () => { expect(onSendError).not.toHaveBeenCalled() }) + + it('reports the takeover once an Escape is accepted', async () => { + await render(true, 'stream-1') + + act(() => stop?.()) + await act(async () => vi.runAllTimersAsync()) + + expect(reportWorkerTerminalUserInput).toHaveBeenCalledWith( + expect.objectContaining({ sendRequest }), + 'terminal-1' + ) + }) + + it('does not report a Stop the host rejected', async () => { + sendRequest.mockResolvedValue({ ok: true, result: { send: { accepted: false } } }) + await render(true, 'stream-1') + + act(() => stop?.()) + await act(async () => vi.runAllTimersAsync()) + + expect(reportWorkerTerminalUserInput).not.toHaveBeenCalled() + }) }) diff --git a/mobile/src/session/use-mobile-native-chat-stop.ts b/mobile/src/session/use-mobile-native-chat-stop.ts index 871d221abb2..69d2d8e73e8 100644 --- a/mobile/src/session/use-mobile-native-chat-stop.ts +++ b/mobile/src/session/use-mobile-native-chat-stop.ts @@ -3,6 +3,7 @@ import type { RpcClient } from '../transport/rpc-client' import { isRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' import { isLogicalClientCutoverError } from '../transport/stable-logical-rpc-client' import { isTerminalSendRpcAccepted } from '../terminal/terminal-send-rpc-response' +import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report' import { openMobileNativeChatSendBudget } from './mobile-native-chat-send' export function useMobileNativeChatStop(args: { @@ -111,6 +112,8 @@ export function useMobileNativeChatStop(args: { .then((response) => { if (isTerminalSendRpcAccepted(response)) { sawAccepted = true + // A deliberate Stop is human input; it takes the worker over like any other key. + reportWorkerTerminalUserInput(client, handle) } else { sawRejected = true } diff --git a/mobile/src/session/use-mobile-session-terminal-input.ts b/mobile/src/session/use-mobile-session-terminal-input.ts index 02906099e79..3f6e417e23a 100644 --- a/mobile/src/session/use-mobile-session-terminal-input.ts +++ b/mobile/src/session/use-mobile-session-terminal-input.ts @@ -1,4 +1,6 @@ +import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report' import { useCallback } from 'react' +import { isTerminalSendRpcAccepted } from '../terminal/terminal-send-rpc-response' import { clearTerminalLiveInputFocusTimer, scheduleTerminalLiveInputFocus @@ -110,7 +112,7 @@ export function useMobileSessionTerminalInput(scope: MobileSessionFileActionsMod terminalGestureInputInFlightRef.current.add(handle) try { // Why: gesture arrows parked across a reconnect would move a TUI long after the swipe. - await rpc.sendRequest( + const response = await rpc.sendRequest( 'terminal.send', buildTerminalSendParams({ terminal: handle, @@ -120,6 +122,9 @@ export function useMobileSessionTerminalInput(scope: MobileSessionFileActionsMod }), TERMINAL_INPUT_SEND_OPTIONS ) + if (isTerminalSendRpcAccepted(response)) { + reportWorkerTerminalUserInput(rpc, handle) + } } catch { // Transient failure } finally { diff --git a/mobile/src/session/use-mobile-session-terminal-send-actions.ts b/mobile/src/session/use-mobile-session-terminal-send-actions.ts index 6909f71ca63..aa365d5ba39 100644 --- a/mobile/src/session/use-mobile-session-terminal-send-actions.ts +++ b/mobile/src/session/use-mobile-session-terminal-send-actions.ts @@ -1,3 +1,4 @@ +import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report' import { useCallback } from 'react' import { Keyboard } from 'react-native' import { triggerError } from '../platform/haptics' @@ -98,6 +99,9 @@ export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminal TERMINAL_INPUT_SEND_OPTIONS ) const accepted = isTerminalSendRpcAccepted(response) + if (accepted) { + reportWorkerTerminalUserInput(client, activeHandle) + } if (!accepted) { restoreRejectedDraft() } @@ -166,7 +170,16 @@ export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminal }), TERMINAL_INPUT_SEND_OPTIONS ) - .then(isTerminalSendRpcAccepted, () => false) + .then( + (response) => { + const accepted = isTerminalSendRpcAccepted(response) + if (accepted) { + reportWorkerTerminalUserInput(rpc, handle) + } + return accepted + }, + () => false + ) }, [showToast] ) diff --git a/mobile/src/session/use-mobile-terminal-paste.ts b/mobile/src/session/use-mobile-terminal-paste.ts index 57ea720c4c8..3680fa2e505 100644 --- a/mobile/src/session/use-mobile-terminal-paste.ts +++ b/mobile/src/session/use-mobile-terminal-paste.ts @@ -1,4 +1,6 @@ +import { reportWorkerTerminalUserInput } from '../terminal/worker-terminal-takeover-report' import { useCallback, type RefObject } from 'react' +import { isTerminalSendRpcAccepted } from '../terminal/terminal-send-rpc-response' import * as Clipboard from 'expo-clipboard' import { File as FsFile, Paths } from 'expo-file-system' import { ImageManipulator, SaveFormat } from 'expo-image-manipulator' @@ -155,7 +157,7 @@ export function useMobileTerminalPaste({ ) { return } - await currentClient.sendRequest('terminal.send', { + const response = await currentClient.sendRequest('terminal.send', { terminal: targetHandle, text: payload, enter: false, @@ -163,6 +165,9 @@ export function useMobileTerminalPaste({ ? { client: { id: deviceTokenRef.current, type: 'mobile' as const } } : {}) }) + if (isTerminalSendRpcAccepted(response)) { + reportWorkerTerminalUserInput(currentClient, targetHandle) + } onSuccess() refreshCanPaste() } catch (e) { diff --git a/mobile/src/terminal/terminal-live-accessory-raw-send.ts b/mobile/src/terminal/terminal-live-accessory-raw-send.ts index 3824d750792..6fa00ce7bac 100644 --- a/mobile/src/terminal/terminal-live-accessory-raw-send.ts +++ b/mobile/src/terminal/terminal-live-accessory-raw-send.ts @@ -1,3 +1,4 @@ +import { reportWorkerTerminalUserInput } from './worker-terminal-takeover-report' import { getTerminalLiveAccessoryRawSendTarget } from './terminal-live-accessory-raw-send-target' import { isTerminalSendRpcAccepted } from './terminal-send-rpc-response' import { buildTerminalSendParams, TERMINAL_INPUT_SEND_OPTIONS } from './terminal-send-request' @@ -37,5 +38,14 @@ export async function sendTerminalLiveAccessoryRawBytes( }), TERMINAL_INPUT_SEND_OPTIONS ) - .then(isTerminalSendRpcAccepted, () => false) + .then( + (response) => { + const accepted = isTerminalSendRpcAccepted(response) + if (accepted) { + reportWorkerTerminalUserInput(args.client!, rawSendTarget) + } + return accepted + }, + () => false + ) } diff --git a/mobile/src/terminal/terminal-webview-payload-hash.test.ts b/mobile/src/terminal/terminal-webview-payload-hash.test.ts index f8bfa4bd134..66cd2735ea2 100644 --- a/mobile/src/terminal/terminal-webview-payload-hash.test.ts +++ b/mobile/src/terminal/terminal-webview-payload-hash.test.ts @@ -6,8 +6,8 @@ import { XTERM_HTML } from './terminal-webview-html' // uncovered region ships silently. A diff here means the emitted WebView source changed — // update these values only when that change is deliberate, and only after checking the // document still runs. Refactors that merely move slice boundaries must leave them alone. -const EXPECTED_SHA256 = '42cc000faddc3b58b8fd4855f848c7878f0cd6166c613f66d733645e8e1b9608' -const EXPECTED_LENGTH = 729776 +const EXPECTED_SHA256 = '5c69dce3236662c381abbfb5d2d6b7163e0f4dd6841d72753733f9470326fee3' +const EXPECTED_LENGTH = 730428 describe('terminal WebView payload', () => { it('composes the expected document', () => { diff --git a/mobile/src/terminal/terminal-webview-theme-injected.test.ts b/mobile/src/terminal/terminal-webview-theme-injected.test.ts index 92e4127b2fc..d0947ef3e92 100644 --- a/mobile/src/terminal/terminal-webview-theme-injected.test.ts +++ b/mobile/src/terminal/terminal-webview-theme-injected.test.ts @@ -76,4 +76,43 @@ describe('mobile terminal-webview contrast floor gate', () => { context.applyTerminalTheme({ theme: { background: '#1e242a' } }) expect(term.options.minimumContrastRatio).toBe(DARK_FLOOR) }) + + // #10754: the desktop user can lower or disable the floor. Mobile mirrors the desktop gate, so the + // published value has to win here or the same session renders differently on the phone. + describe('published desktop override', () => { + function applyOn(term: { options: { minimumContrastRatio: number } }, input: unknown): void { + const context = loadThemeInjected({ + term, + document: { + documentElement: { style: { background: '' } }, + body: { style: { background: '' } } + } + }) as Record & { applyTerminalTheme: (input: unknown) => void } + context.applyTerminalTheme(input) + } + + it('uses the published floor instead of the luminance gate', () => { + const term = { options: { minimumContrastRatio: 0 } } + applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: 1 }) + expect(term.options.minimumContrastRatio).toBe(1) + }) + + it("clamps a published floor to xterm's 1-21 window", () => { + const term = { options: { minimumContrastRatio: 0 } } + applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: 99 }) + expect(term.options.minimumContrastRatio).toBe(21) + applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: 0 }) + expect(term.options.minimumContrastRatio).toBe(1) + }) + + it('falls back to the luminance gate for an older host that omits the field', () => { + const term = { options: { minimumContrastRatio: 0 } } + for (const published of [undefined, null, 'off', Number.NaN]) { + applyOn(term, { theme: { background: '#1e242a' }, minimumContrastRatio: published }) + expect(term.options.minimumContrastRatio).toBe(DARK_FLOOR) + applyOn(term, { theme: { background: '#ffffff' }, minimumContrastRatio: published }) + expect(term.options.minimumContrastRatio).toBe(LIGHT_FLOOR) + } + }) + }) }) diff --git a/mobile/src/terminal/terminal-webview-theme-injected.ts b/mobile/src/terminal/terminal-webview-theme-injected.ts index c2d9beb4786..98489b219c7 100644 --- a/mobile/src/terminal/terminal-webview-theme-injected.ts +++ b/mobile/src/terminal/terminal-webview-theme-injected.ts @@ -5,7 +5,8 @@ import { colors } from '../theme/mobile-theme' // #7934/#10104): a dark composed background gets a mild floor of 3 to rescue near-background body text // (e.g. Antigravity's #262b30 on #1e242a) without over-brightening vibrant ANSI colors; a light // background keeps the WCAG-AA 4.5 floor. Gate on the composed background luminance, not app mode, -// because either theme slot can hold either kind of theme. +// because either theme slot can hold either kind of theme. An explicit desktop override published on +// the theme payload (#10754) wins over the luminance gate; older hosts simply omit it. export const TERMINAL_WEBVIEW_THEME_JS = ` var DARK_BG_MIN_CONTRAST = 3; var LIGHT_BG_MIN_CONTRAST = 4.5; @@ -63,6 +64,12 @@ export const TERMINAL_WEBVIEW_THEME_JS = ` return (Math.max(la, lb) + 0.05) / (Math.min(la, lb) + 0.05); } + // Clamp an explicit desktop override to xterm's 1-21 range; null means "no usable override". + function normalizeTerminalContrastOverride(value) { + if (typeof value !== 'number' || !isFinite(value)) return null; + return Math.min(21, Math.max(1, value)); + } + // Pick the xterm minimumContrastRatio floor from the composed terminal background. // Unparseable input defaults to the dark floor so agent output never stays invisible. function resolveTerminalContrastFloor(background) { @@ -100,7 +107,13 @@ export const TERMINAL_WEBVIEW_THEME_JS = ` var background = terminalTheme.background || '${colors.terminalBg}'; document.documentElement.style.background = background; document.body.style.background = background; - terminalMinimumContrastRatio = resolveTerminalContrastFloor(background); + // Why prefer the published value: the desktop user may have lowered or disabled the floor (#10754); + // an older host omits the field and the luminance gate stays authoritative. + var publishedFloor = normalizeTerminalContrastOverride( + input && typeof input === 'object' ? input.minimumContrastRatio : undefined + ); + terminalMinimumContrastRatio = + publishedFloor === null ? resolveTerminalContrastFloor(background) : publishedFloor; if (term) { term.options.theme = terminalTheme; term.options.minimumContrastRatio = terminalMinimumContrastRatio; diff --git a/mobile/src/terminal/worker-terminal-takeover-report.test.ts b/mobile/src/terminal/worker-terminal-takeover-report.test.ts new file mode 100644 index 00000000000..5ef62be1f0b --- /dev/null +++ b/mobile/src/terminal/worker-terminal-takeover-report.test.ts @@ -0,0 +1,82 @@ +import { beforeEach, afterEach, expect, it, vi } from 'vitest' +import { + reportWorkerTerminalUserInput, + resetWorkerTerminalTakeoverReportsForTest +} from './worker-terminal-takeover-report' + +const success = { id: 'report', ok: true as const, result: { changed: 1 } } +beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(1_000) + resetWorkerTerminalTakeoverReportsForTest() +}) +afterEach(() => vi.useRealTimers()) + +it('gates per handle and owning client for 30 seconds', () => { + const relay = { sendRequest: vi.fn().mockResolvedValue(success) } + const direct = { sendRequest: vi.fn().mockResolvedValue(success) } + for (let i = 0; i < 100; i++) { + reportWorkerTerminalUserInput(relay, 'term-1') + } + expect(relay.sendRequest).toHaveBeenCalledTimes(1) + reportWorkerTerminalUserInput(relay, 'term-2') + reportWorkerTerminalUserInput(direct, 'term-1') + expect(relay.sendRequest).toHaveBeenCalledTimes(2) + expect(direct.sendRequest).toHaveBeenCalledTimes(1) + vi.advanceTimersByTime(29_999) + reportWorkerTerminalUserInput(relay, 'term-1') + expect(relay.sendRequest).toHaveBeenCalledTimes(2) + vi.advanceTimersByTime(1) + reportWorkerTerminalUserInput(relay, 'term-1') + expect(relay.sendRequest).toHaveBeenCalledTimes(3) + expect(relay.sendRequest).toHaveBeenLastCalledWith( + 'orchestration.workerTerminalUserInput', + { terminal: 'term-1' }, + { timeoutMs: 5_000, budgetSpansConnect: true, failWhenDisconnected: true } + ) +}) + +it('does not await a report and coalesces input while it is pending', () => { + const client = { sendRequest: vi.fn(() => new Promise(() => {})) } + expect(reportWorkerTerminalUserInput(client, 'term-1')).toBeUndefined() + reportWorkerTerminalUserInput(client, 'term-1') + expect(client.sendRequest).toHaveBeenCalledTimes(1) +}) + +it.each(['throw', 'rpc refusal'])( + 'retries a %s once on the same target, then permits a later attempt', + async (failure) => { + const client = { + sendRequest: + failure === 'throw' + ? vi.fn().mockRejectedValue(new Error('offline')) + : vi.fn().mockResolvedValue({ id: 'report', ok: false, error: { message: 'refused' } }) + } + reportWorkerTerminalUserInput(client, 'term-1') + await vi.advanceTimersByTimeAsync(249) + expect(client.sendRequest).toHaveBeenCalledTimes(1) + await vi.advanceTimersByTimeAsync(1) + expect(client.sendRequest).toHaveBeenCalledTimes(2) + await vi.advanceTimersByTimeAsync(1_000) + expect(client.sendRequest).toHaveBeenCalledTimes(2) + client.sendRequest.mockResolvedValue(success) + reportWorkerTerminalUserInput(client, 'term-1') + expect(client.sendRequest).toHaveBeenCalledTimes(3) + } +) + +it('a report that changed nothing still arms the gate, so plain terminals pay once per window', async () => { + // Why: the host answers `changed: 0` for every ordinary terminal; reopening on that turned + // every accepted key into an RPC and a host write transaction (round 6 measurement: 100 for 100). + const client = { + sendRequest: vi.fn().mockResolvedValue({ id: 'report', ok: true, result: { changed: 0 } }) + } + for (let i = 0; i < 100; i++) { + reportWorkerTerminalUserInput(client, 'term-plain') + await vi.advanceTimersByTimeAsync(100) + } + expect(client.sendRequest).toHaveBeenCalledTimes(1) + await vi.advanceTimersByTimeAsync(30_000) + reportWorkerTerminalUserInput(client, 'term-plain') + expect(client.sendRequest).toHaveBeenCalledTimes(2) +}) diff --git a/mobile/src/terminal/worker-terminal-takeover-report.ts b/mobile/src/terminal/worker-terminal-takeover-report.ts new file mode 100644 index 00000000000..a3ed7f6424d --- /dev/null +++ b/mobile/src/terminal/worker-terminal-takeover-report.ts @@ -0,0 +1,59 @@ +import type { RpcClient } from '../transport/rpc-client' + +type ReportClient = Pick +const REPORT_INTERVAL_MS = 30_000 +const REPORT_RETRY_DELAY_MS = 250 +let reportsByClient = new WeakMap>() + +// The same logical client owns relay/direct cutover; never reroute a report via active UI state. +export function reportWorkerTerminalUserInput(client: ReportClient, terminal: string): void { + let reports = reportsByClient.get(client) + if (!reports) { + reports = new Map() + reportsByClient.set(client, reports) + } + const now = Date.now() + const last = reports.get(terminal) + if (last !== undefined && now - last < REPORT_INTERVAL_MS) { + return + } + if (reports.size >= 256) { + for (const [handle, reportedAt] of reports) { + if (now - reportedAt >= REPORT_INTERVAL_MS) { + reports.delete(handle) + } + } + } + // Why the gate ignores the answer: like desktop, one report per terminal per window is the + // whole cost of typing into any terminal, worker or not. A result-aware gate that reopened on + // "changed nothing" turned every key on an ordinary terminal into an RPC plus a host write. + reports.set(terminal, now) + void sendTakeoverReport(client, terminal).catch(() => { + if (reports.get(terminal) === now) { + reports.delete(terminal) + } + }) +} + +async function sendTakeoverReport(client: ReportClient, terminal: string): Promise { + const report = async (): Promise => { + const response = await client.sendRequest( + 'orchestration.workerTerminalUserInput', + { terminal }, + { timeoutMs: 5_000, budgetSpansConnect: true, failWhenDisconnected: true } + ) + if (!response.ok) { + throw new Error('Worker takeover report rejected') + } + } + try { + return await report() + } catch { + await new Promise((resolve) => setTimeout(resolve, REPORT_RETRY_DELAY_MS)) + return await report() + } +} + +export function resetWorkerTerminalTakeoverReportsForTest(): void { + reportsByClient = new WeakMap() +} diff --git a/package.json b/package.json index 9d42e149ad4..0536f4fd606 100644 --- a/package.json +++ b/package.json @@ -134,6 +134,7 @@ "test:e2e:terminal-ime-native": "node config/scripts/run-terminal-ibus-hangul-e2e.mjs", "test:e2e:computer": "vitest run --config tests/e2e/vitest.config.ts", "bench:idle-cpu": "pnpm run ensure:electron-runtime && node config/scripts/run-idle-cpu-benchmark.mjs", + "bench:spinners": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/spinner-rendering/run.mjs", "bench:macos-computer-helper-owner-loss": "node config/scripts/macos-computer-helper-owner-loss-benchmark.mjs", "bench:startup": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/startup-time-bench.mjs", "bench:daemon-coldstart": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/daemon-coldstart-bench.mjs", @@ -236,12 +237,13 @@ "clsx": "^2.1.1", "cmdk": "^1.1.1", "dompurify": "3.4.14", - "electron": "^43.4.1", + "electron": "43.6.0", "electron-builder": "^26.15.3", "electron-builder-squirrel-windows": "^26.15.3", "electron-vite": "^5.0.0", "emoji-picker-react": "^4.19.1", "emojibase-data": "17.0.0", + "esbuild": "^0.25.12", "happy-dom": "^20.11.8", "html-to-image": "^1.11.13", "husky": "^9.1.7", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 02f2671ad7a..7add63b397b 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -127,10 +127,10 @@ importers: version: 0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4) '@electron-toolkit/preload': specifier: ^3.0.2 - version: 3.0.2(electron@43.4.1(supports-color@7.2.0)) + version: 3.0.2(electron@43.6.0(supports-color@7.2.0)) '@electron-toolkit/utils': specifier: ^4.0.0 - version: 4.0.0(electron@43.4.1(supports-color@7.2.0)) + version: 4.0.0(electron@43.6.0(supports-color@7.2.0)) '@floating-ui/dom': specifier: 1.7.6 version: 1.7.6 @@ -349,8 +349,8 @@ importers: specifier: 3.4.14 version: 3.4.14 electron: - specifier: ^43.4.1 - version: 43.4.1(supports-color@7.2.0) + specifier: 43.6.0 + version: 43.6.0(supports-color@7.2.0) electron-builder: specifier: ^26.15.3 version: 26.15.3(electron-builder-squirrel-windows@26.15.3) @@ -366,6 +366,9 @@ importers: emojibase-data: specifier: 17.0.0 version: 17.0.0(emojibase@17.0.0) + esbuild: + specifier: ^0.25.12 + version: 0.25.12 happy-dom: specifier: ^20.11.8 version: 20.11.8 @@ -4327,8 +4330,8 @@ packages: resolution: {integrity: sha512-bO3y10YikuUwUuDUQRM4KfwNkKhnpVO7IPdbsrejwN9/AABJzzTQ4GeHwyzNSrVO+tEH3/Np255a3sVZpZDjvg==} engines: {node: '>=8.0.0'} - electron@43.4.1: - resolution: {integrity: sha512-5b+EuiwkgG5iRcsEL34rimgRpkYp15SsfZOa0pC5kXs0Tb82TH4n95rpQzTZa7yRCbA7tm0WoEbuBL6NaAhAcA==} + electron@43.6.0: + resolution: {integrity: sha512-DqVKYV+FXheMSLTxcMQ+NCo78BDgpnToSyIzXctlUtbP3lRGEuoo1P+C2n/90rJ7TvHgzP0bpP9fbbXxp4noIg==} engines: {node: '>= 22.12.0'} hasBin: true @@ -7329,17 +7332,17 @@ snapshots: '@electron-internal/extract-zip@1.0.4': {} - '@electron-toolkit/preload@3.0.2(electron@43.4.1(supports-color@7.2.0))': + '@electron-toolkit/preload@3.0.2(electron@43.6.0(supports-color@7.2.0))': dependencies: - electron: 43.4.1(supports-color@7.2.0) + electron: 43.6.0(supports-color@7.2.0) '@electron-toolkit/tsconfig@2.0.0(@types/node@25.9.5)': dependencies: '@types/node': 25.9.5 - '@electron-toolkit/utils@4.0.0(electron@43.4.1(supports-color@7.2.0))': + '@electron-toolkit/utils@4.0.0(electron@43.6.0(supports-color@7.2.0))': dependencies: - electron: 43.4.1(supports-color@7.2.0) + electron: 43.6.0(supports-color@7.2.0) '@electron/asar@3.4.1': dependencies: @@ -10685,7 +10688,7 @@ snapshots: transitivePeerDependencies: - supports-color - electron@43.4.1(supports-color@7.2.0): + electron@43.6.0(supports-color@7.2.0): dependencies: '@electron-internal/extract-zip': 1.0.4 '@electron/get': 5.0.0(supports-color@7.2.0) diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md index 4e49a0d84af..d744785ab10 100644 --- a/skill-guides/orchestration.md +++ b/skill-guides/orchestration.md @@ -137,8 +137,8 @@ After three consecutive empty waits, stop waiting blindly and enumerate with `ORCA orchestration worker-list --include-remote --json` (defaults to the bound Run; `--run ` overrides; the receipt's `scope` names which), acting on each row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv. -An `inspect` `nextAction` on a `live` row with `attention.requiresAction` false -is informational, not a command to re-run: keep waiting with `check --wait`. +A `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting +with `check --wait`. Absence never earns an argv; settlement and pending work still do. Leave the wait only on positive proof the agent stopped: `exited` liveness, the worker's own observation of process exit, or a transcript whose final agent turn sent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index fc30604a264..47b5559f01b 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -66,10 +66,10 @@ const ORCA_PER_WORKSPACE_ENV_SSH_HOST_REFERENCE_MARKDOWN = "# SSH connection mod const ORCA_PER_WORKSPACE_ENV_WINDOWS_SCRIPTS_REFERENCE_MARKDOWN = "# Windows local-side scripts\n\nLoad this when the user's desktop is Windows and you are scaffolding the local-side scripts. A bare\n`.sh` will not execute there. Either require WSL or Git Bash and point `orca.yaml` at a launcher such\nas `bash ./scripts/orca-vm/.sh` through a `.cmd` file, or scaffold PowerShell equivalents.\n\nThe remote-side commands you run inside the Linux environment stay bash regardless of the desktop OS.\n\n```powershell\n#requires -Version 5\n$ErrorActionPreference = 'Stop'\n# resolve env→state→fallback; run the provider CLI / ssh the same way;\n# capture provider output; build the result object for the chosen mode and write ONE line of JSON to stdout.\n# Orca-server mode: @{ schemaVersion=1; pairingCode=$pairingCode; projectRoot=$projectRoot; userData=@{...} }\n# SSH mode: @{ schemaVersion=1; connection=@{ type=\"ssh\"; projectRoot=$projectRoot;\n# target=@{ label=$label; host=$host; port=$port; username=$user } } }\n($result | ConvertTo-Json -Compress -Depth 6)\n# progress/errors → Write-Error / the error stream, never stdout.\n```\n\nThe doctor's executable-bit check is a POSIX concept and is skipped on Windows, so a script that is\nunusable on the user's machine for a different reason still has to be caught by the `--provision`\nself-test.\n" // oxfmt-ignore -const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nAn `inspect` `nextAction` on a `live` row with `attention.requiresAction` false\nis informational, not a command to re-run: keep waiting with `check --wait`.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n" +const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nA `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting\nwith `check --wait`. Absence never earns an argv; settlement and pending work still do.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n" // oxfmt-ignore -const ORCHESTRATION_FULL_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nAn `inspect` `nextAction` on a `live` row with `attention.requiresAction` false\nis informational, not a command to re-run: keep waiting with `check --wait`.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n\n---\n\n# Bundled references\n\nThese references belong to the version-matched guide above. Read only the documents named by its action gates.\n\n\n\n# Coordinator loop\n\nLoad this reference for expanded DAG waves, per-invocation launch preferences,\nsame-terminal reuse, or review ownership. The compact guide remains the source\nof truth for the loop order and completion boundary.\n\n## Ready waves\n\nCreate independent Tasks before the first wait. Encode only real dependencies,\nthen use the ready view as external memory:\n\n```text\nORCA orchestration task-create --spec \"\" --deps --json\nORCA orchestration task-list --ready --brief --json\n```\n\n`--brief` collapses whitespace and caps echoed specs at 160 characters;\n`spec_truncated` identifies shortened rows. Omit it when full specs are needed or\nwhen an older CLI rejects the flag. A nested worker must respect\n`nested_worker_depth_exceeded`; creating another Run does not reset depth.\n\n## Launch preferences\n\nFor a fresh Claude, Codex, or Cursor terminal, `--model` accepts an opaque\nprovider model ID. Pass it only when the user named a model; otherwise omit it\nso the worker inherits the user's configured agent default. Add `--effort` only\nwhen that model supports it:\n\n```text\nORCA orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`; neither option combines with `--terminal`. A\nconnected worker server must advertise launch-preference support before Orca\nforwards either field. Compare `launch.requested` with `launch.effective`; never\nclaim a model or effort from requested arguments alone.\n\n## Reuse after settlement\n\nChoose the terminal's next owner before acknowledging the Delivery. When the\nsame exact agent has immediate follow-up work, recover the proven handle and\ntransfer cleanup ownership to the new Dispatch:\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-start --task --terminal --json\n```\n\nOtherwise explicitly retain or release the settled worker. Do not leave it live\nonly to inspect output; archived output remains available through `worker-read`.\n\n## Review ownership\n\nA review-only `worker_done` authorizes synthesis of findings, not coordinator\nfile edits. Dispatch or hand off fixes unless the user explicitly assigned them\nto the coordinator. If the user's plan names a next owner, post-review fixes and\nPR preparation remain with that owner; the coordinator routes and synthesizes.\n\n\n\n# Legacy contract migration\n\nLoad this reference only for an authority label, adopted Run, compatibility or\nrecovery receipt, or explicit legacy takeover. A newly created attempt always\nuses the current grammar.\n\n## Authority labels\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported\n command printed with the message, using the same selected executable and\n arguments supplied by the original prompt.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded,\n at-least-once cutover replay. Process it idempotently and acknowledge only\n through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or\n lifecycle mutation.\n- An unlabeled current message uses the current guide and grammar.\n\nAn explicitly selected current Run, attested current binding, current Dispatch,\nor federated attachment takes precedence over legacy fallback. A retained\nadoption record alone does not grant mutation authority. If liveness, principal\nownership, capability, or the exact legacy contract is unproven, degrade to\nread-only inspection and never fall back to local execution.\n\nAdoption preserves the live agent process, PTY/session, terminal handle,\ntab/pane, worktree or folder workspace, Task, and Dispatch. It never restarts or\nreplaces the worker and never revives the retired scheduler. Loss of lifecycle\nauthority does not invalidate the existing process, assignment, or filesystem\nwork. Exact recovery may restore the same PTY once in its original inactive\nbackground tab; it must not spawn, write, signal, stop, switch, focus, split, or\ninject a terminal.\n\n## Compatibility recovery\n\nWhen a compatibility response returns structured next-step arguments, execute\nthose exact arguments with the same selected CLI executable. Do not translate\nfrom memory, broaden the recipient, or retry as a current mutation unless the\nreceipt explicitly authorizes it.\n\nA pending ask, reply, final Dispatch settlement, and consuming check have\ndurable recovery identities. Heartbeat and escalation remain at-least-once\nacross a manual contract-boundary retry. If an ask may already have been\nanswered, run the exact non-consuming recovery check printed by Orca before\ncreating any new question. Never guess among identical question threads.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The\ninitial command commits the question, prints its exact\n`ask --resume ` command, and exits with launcher status `75`. Run\nthat exact resume after the launcher or update boundary. For an attested WSL\nlaunch, preserve the printed `orca-ide` executable and distro route. Older WSL\nworkers without launch proof remain lifecycle read-only even while their\nterminal and filesystem work continue.\n\n## Read-only inspection and takeover\n\nRead-only inspection does not consume mail:\n\n```text\nORCA orchestration run-list --json\nORCA orchestration run-show --id run_legacy_local --json\nORCA orchestration run-show --id --json\nORCA orchestration task-list --run --json\nORCA orchestration inbox --full --json\nORCA orchestration check --terminal --peek --format --json\nORCA terminal read --terminal --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\n`run_legacy_local` is an empty audit tombstone after adoption. Find the ordinary\nRun whose objective is `Recovered orchestration work from a contract update`.\n\nOnly when the original coordinator is unavailable or cannot prove retained\nauthority may a new live coordinator take over from its own terminal:\n\n```text\nORCA orchestration run-use --id --takeover-legacy --json\nORCA orchestration check --run --json\n```\n\nTakeover binds the authenticated invoking terminal; `--from` cannot nominate\nanother coordinator. It fences only the old coordinator and moves pending mail\ninto current Run delivery. It preserves live workers, Tasks, Dispatches, processes, and files.\nNever take over while the original coordinator is actively coordinating.\n\nDo not launch a replacement editor merely because Orca updated or authority is\nunclear. Keep the original worker as the only editor until a stable handoff\npoint, then use a fresh current Dispatch in a conflict-free placement.\n\n\n\n# Low-level topology\n\nLoad this reference only when `worker-start` cannot express required custom argv\nor terminal topology. It is not the normal supervised loop and is never a full\nhandoff recipe.\n\n```text\nORCA terminal create --worktree active --title --command \"\" --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\nORCA orchestration dispatch --task --to --inject --json\n```\n\nWait for readiness only when startup could lose injected input. Prefer\nagent-first `worker-start` whenever its argv and topology are sufficient.\n\n`dispatch --inject` creates authoritative Task/Dispatch context but deliberately\nkeeps an operator-created process unsupervised: it creates no supervised worker\nresource row. `worker-show`, `worker-read`, and `worker-list` report the lane as\n`unsupervised`; `worker-stop` and `worker-abandon` do not close that process, and\nsettled retain/release take no process action.\n\nUse `worker-start --terminal ` when lifecycle ownership of an existing\nagent terminal is required. Never imply that low-level dispatch retroactively\nowns a process, never use it to route around the nested-depth limit, and never\nuse it for an ownership handoff.\n\n\n\n# Messaging and gates\n\nLoad this reference for inbox replay, attempt-specific guidance, group\naddresses, blocking questions, or coordinator-managed DAG decisions.\n\nA successful `send` proves durable enqueue. Wake and nudge are best-effort\nattention only: neither proves the recipient read the message, began a turn, or\naccepted steering.\n\n## Coordinator delivery loop\n\n`check` names its caller with `--terminal ` and is the only verb that\nrejects `--from`. Omit `--terminal` inside an Orca terminal, where Orca resolves\nthe caller; pass it explicitly from anywhere else, including a dispatched\nworker reading coordinator follow-ups.\n\nA consuming coordinator `check` returns the bound Run's oldest FIFO Delivery,\nup to 50 messages, and replays that exact batch until acknowledged. Process\nevery row and required terminal ownership decision before `--ack`. Type filters\ndecide when a waiter wakes; they do not authorize skipping older actionable\nmail. A Delivery therefore always carries the whole FIFO batch whatever its\ntypes, and a `check` without `--wait` hands that batch over unfiltered.\n`--peek` and `--all` are read-only inspection, not progress through the\ncoordinator inbox.\n\nAn empty wait or timeout is a checkpoint. Continue rolling waits until every\nexpected Dispatch settles. Heartbeat or visible activity means alive, not done.\n\n## Addresses\n\nUse a stable Dispatch address for attempt-specific coordinator guidance:\n\n```text\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nDo not substitute a remote terminal handle. Omit `--from` for ordinary\ncoordinator calls; a dispatched worker instead copies the exact `--from` and\ncapability arguments in its preamble. `check` is the exception: it identifies\nits caller with `--terminal`, never `--from`.\n\nGroup addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`,\n`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Use them only for\nintentional fan-out status or questions. `worker_done`, heartbeat, and other\nDispatch lifecycle messages never target groups.\n\n## Questions and gates\n\nA worker uses `ask`; its timeout leaves one durable question pending, which the\nworker resumes by message ID. The coordinator answers that message with `reply`.\n\nUse a gate only for a coordinator-owned Task-DAG decision:\n\n```text\nORCA orchestration gate-create --task --question \"\" --options --json\nORCA orchestration gate-resolve --id --resolution \"\" --json\nORCA orchestration gate-list --task --json\n```\n\nPass `json_array` using the quoting rules of the active shell; do not copy POSIX\nsingle-quote syntax into PowerShell or `cmd.exe`.\n\nDo not create a gate merely to answer a worker's `ask`.\n\n\n\n# Placement and remote execution\n\nLoad this reference before creating a new worktree or placing work through SSH,\nWSL, or another connected Orca server.\n\n## Placement choices\n\nA fresh worker means a fresh agent terminal, not a new Git worktree. Use the\ncurrent or an exact existing workspace by default. Create a worktree only when\nthe user requested one or a concrete checkout or filesystem conflict makes\nsharing unsafe.\n\n```text\n# Current workspace; setup is not rerun.\nORCA orchestration worker-start --task --worktree current --agent codex --json\n\n# Stacked child worktree.\nORCA orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n\n# Independent top-level worktree.\nORCA orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nCurrent and exact existing workspaces create a fresh terminal unless\n`--terminal` is explicit. Folder workspaces are first-class; do not invoke Git\nor require worktree lineage when the selected workspace is a folder.\n\nRegister a folder workspace through project setup. `repo add --path `\nrequires a valid Git repository and rejects a plain directory:\n\n```text\nORCA project setup-existing-folder --project --host --path --kind folder --json\n```\n\nThen place work on the returned workspace with an exact selector. A worktree\nselector needs the full `::` value Orca returned, passed as\n`id:`; a bare repo id is not a worktree id. `new-child` and\n`new-top-level` are worktree creation and do not apply to a folder.\n\nNew worktrees use agent-first creation and run setup by default. Preserve the\nrepository's startup policy: `start-immediately` can report setup as `running`,\nwhile `wait-for-setup` gates prompt delivery on success. Orca lineage, Git base,\nfilesystem isolation, coordination parentage, UI grouping, and execution host\nare separate decisions.\n\n## Connected servers\n\nThe Run and Tasks remain authoritative on the current server. `--on` selects\nonly the worker's execution server and appears only on `worker-start`:\n\n```text\nORCA orchestration worker-start --task --on --worktree new-top-level --repo --name --agent codex --setup run --json\n```\n\nRemote `current` and `new-child` are invalid because they are ambiguous across\nservers. Use an exact discovered remote workspace, or `new-top-level` with an\nexact remote repository selector. After start, route every follow-up, read,\nstop, and cleanup by Dispatch ID; never repeat `--on` or substitute a remote\nterminal handle.\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\nORCA orchestration worker-list --run --include-remote --json\n```\n\n`worker-list` reads local fleet state only; enumerate remote workers with\n`--include-remote` or every one of them reads `unverifiable`. Scope every list\nwith `--run `: unscoped, it reports every Dispatch this runtime has\nrecorded, and the workers you are waiting on are lost in that history.\n\n## Execution-host and mixed-version floor\n\nThe execution host owns process, filesystem, transcript, stop, and cleanup\nfacts. Render only `live`, `unverifiable`, or `exited`. Connection loss, relay\nabsence, missing client inventory, or timeout yields `unverifiable`, never\nsynthetic exit and never a client-local substitute action.\n\nClients and servers update independently. Optional response fields may be\nabsent. Forward model/effort, transcript reads, cleanup, or another new remote\noperation only when the peer advertises the relevant capability; unknown stream\nopcodes can be silently dropped. A narrow unsupported response may degrade to a\ndocumented older path, but must not broaden the target or cross the execution\nboundary. Changing host-published content reaches old clients even without a\nwire-shape change, so preserve established semantics or negotiate the behavior.\n\nFor WSL, use the exact executable and arguments returned by Orca so the distro\nand packaged launcher remain bound. Do not translate a printed `orca-ide`\nrecovery command into a PATH-resolved local command.\n\n\n\n# Recovery and cleanup\n\nLoad this reference only after a failed/stopped/unknown attempt, explicit retry\ndecision, stop/abandon request, retention request, or uncertain release.\n\n| Proven state | Safe action |\n| ----------------------- | ------------------------------------------------------------------ |\n| `ready` or active | Keep waiting; optionally read bounded output |\n| `failed` or `stopped` | Start a replacement with `--retry-of`; repeat placement explicitly |\n| `outcome_unknown` | Inspect, then choose `worker-stop` or explicit `worker-abandon` |\n| Accepted `worker_done` | Reuse, retain, or release |\n| Remote contact lost | Preserve `unverifiable`; do not stop or retry from absence alone |\n| `unverifiable` liveness | Keep waiting or inspect; never stop, abandon, retry, or release |\n| Proven `exited` agent | Enumerate with `worker-list`; follow its `nextAction` |\n\n## Inspect before acting\n\n```text\nORCA orchestration worker-list --run --json\nORCA orchestration worker-list --run --include-remote --json\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\n```\n\n`worker-list` is the enumerating command and the authority on agent liveness:\neach row carries `projection.liveness`, `projection.attention.categories`,\n`projection.attention.requiresAction`, and a literal `projection.nextAction`\nargv to run. Always scope it with `--run `; an unscoped list reports\nevery Dispatch this runtime has ever recorded and buries the live ones.\n`worker-show`'s `observation.status` is PTY liveness only, so a `live` terminal\nwhose agent died at a trust prompt still reads `live` there.\n\nWhen the two disagree, the fleet verdict decides — unless the fleet row is\n`unverifiable` for a reason that names a gap on this client rather than a fact\nabout the worker. `missing_status`, `host_unavailable`, and\n`capability_unsupported` are such gaps: the first means this runtime holds no\nstatus row, the second that it could not ask the execution host at all, and the\nthird that a stale peer answered but lacks the fleet-snapshot capability.\nAgainst any of them, a `worker-show` verdict sourced from the execution host is\nthe better evidence and outranks the row. Only `host_unavailable` is contact\nloss; the other two mean the host was never asked or answered without the\ncapability.\n\nThis never promotes absence. `unverifiable` from either command still authorizes\nnothing — only a positive `live` or `exited` verdict does.\n\nA worker started with `--on ` reads `unverifiable` until you\nenumerate with `--include-remote`, which asks its execution host for the\nverdict. Past 100 rows the response pages, so follow `page.nextCursor` with\n`--cursor ` until `page.hasMore` is false.\n\n## Stall needs positive evidence\n\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Only then choose `worker-stop` or `worker-abandon`.\n\n`unverifiable` is always absence — `missing_status`, `stale_status`,\n`restored_unconfirmed`, or a remote worker with no connection — and a null\n`agentWait` or an unchanged `worker-read` tail is that same absence seen again.\nAbsence never authorizes stop, abandon, retry, or release: keep waiting, or\ninspect until you hold one of the positive signals above. A `nextAction` that\nnames an inspecting command is asking for evidence, not for cleanup.\n\n`worker-read --source auto` uses a proven provider transcript when available and\notherwise returns bounded terminal output with a typed `fallbackReason`.\nContinue with its top-level cursor, which is pinned to that source. If Orca\nreports `source_changed`, restart without the old cursor. A bounded initial\ntranscript tail can return an EOF cursor that follows only newly appended records;\nread `contentComplete`, `clipping`, and `warnings` before assuming omitted older\nrecords are pageable. Never guess a provider session ID, transcript path, or\nremote terminal handle.\n\n## Was the mutation applied?\n\nWhen a mutation's response was lost and named no Dispatch, do not replay blind.\nEvery orchestration mutation accepts `--retry-request `, which reuses one\noperation identity so Orca can replay, join, or recover it instead of starting a\nduplicate. Ask what happened first:\n\n```text\nORCA orchestration request-show --request --json\n```\n\n`completed` means the mutation already took effect; read its recorded receipt\ninstead of rerunning. `pending` means the original mutation is still running or\nOrca restarted before recording its outcome; replay the original command with\n`--retry-request `. `absent` means this runtime holds no receipt\nunder your caller identity — that is not proof nothing happened, so inspect the\naffected Task, Dispatch, and terminal before deciding whether to retry.\n\nWhen a worker's terminal accepted input but the submit is unconfirmed, use\n`terminal send --wait-submit `: it observes the accepted prompt for that\nlong and, on timeout, returns the input-accepted receipt without resending.\n\n## Refused starts\n\n`dispatch` and `worker-start` refuse the following preflight cases with a stable\n`error.code`; read it before choosing a recovery, and treat `error.data.nextSteps`\nas the exact recovery text. Older hosts may omit `data`, so treat every field as\noptional.\n\n| Code | Meaning | Recovery |\n| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |\n| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |\n| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |\n| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |\n\n## Retry, stop, and abandon\n\nRetry only a positively proven failed or stopped attempt. Name the failed Task\nwith `--task`, since `--spec` creates a new one. Placement is never silently\ninherited:\n\n```text\nORCA orchestration worker-start --task --retry-of --worktree --agent --json\n```\n\nAfter three consecutive failures for one Task, its dispatch context\ncircuit-breaks and the Task is failed. Do not route around that boundary with a\nnew Run or an unrelated Dispatch.\n\nFor `outcome_unknown`, inspect first, then make an explicit choice:\n\n```text\nORCA orchestration worker-stop --dispatch --json\nORCA orchestration worker-abandon --dispatch --json\n```\n\n`worker-stop` closes only the exact proven supervised agent terminal. It never\ndeletes the worktree, setup terminal, configured tabs, or unrelated processes.\n`worker-abandon` fences orchestration while accepting that resources may remain\nlive; it performs no remote, process, or filesystem action.\n\n## Retain and release\n\n```text\nORCA orchestration worker-retain --dispatch --json\nORCA orchestration worker-release --dispatch --json\n```\n\nRetain only when the user explicitly wants the settled terminal kept live.\nRelease works after succeeded and failed reports, archives readable output, and\ncloses only the exact terminal owned by that settled Dispatch. Replays may call\nrelease again safely. Reused, pre-existing, setup, coordinator, active,\nuser-taken-over, and unproven terminals are retained.\n\nA `worker-start` that failed before its agent was ready still owns the terminal\nit created. Its receipt names `worker-release`, and `worker-list` reports that\nrow as `reclaimable`; release it there rather than closing the terminal by hand.\n\nNever release because of timeout, TUI idle, heartbeat, status, question,\nescalation, or stale/rejected completion. If the receipt says `release_pending`\nor `release_unknown`, follow its exact recovery action. Never substitute\n`terminal close`.\n\n`orchestration reset` is destructive recovery. Do not run it during active\ncoordination unless the user explicitly abandons that state.\n\n\n\n# Worker contract\n\nThe injected preamble is authoritative. Copy its command rather than\nreconstructing flags. In particular, preserve the exact executable, worker\nhandle, Dispatch capability, Task ID, and Dispatch ID.\n\n## Heartbeat\n\nSend heartbeats only at the cadence required by the live preamble. Skip them\nwhile blocked inside `ask` or `check --wait`; those calls are liveness signals.\n\n```text\nORCA orchestration send --from --dispatch-capability --type heartbeat --subject \"alive\" --task-id --dispatch-id --phase \"\"\n```\n\nUse typed lifecycle flags, not a hand-written JSON payload. A heartbeat proves\nliveness, never completion.\n\n## Ask and resume\n\nUse Orca `ask` whenever the coordinator must answer. Never open a local question\nTUI the coordinator cannot answer.\n\n```text\nORCA orchestration ask --from --dispatch-capability --question \"\" --options \",\" --timeout-ms 600000\n\nORCA orchestration ask --from --dispatch-capability --resume --timeout-ms 600000\n```\n\nA timeout or disconnect leaves the original question pending. Resume its\nmessage ID; do not create a duplicate question.\n\n## Reading coordinator follow-ups\n\nThe coordinator steers a running worker with `send --to dispatch:`. That\nenqueue is durable but does not interrupt you, so nothing arrives unless you\nlook:\n\n```text\nORCA orchestration check --terminal --json\n```\n\nRun it at each natural checkpoint — before starting a new file, after a test\nrun — and once more immediately before `worker_done`, so a redirect or a\ncancellation lands before the Task settles. `check` names its caller with\n`--terminal`, never `--from`. Stop checking after `worker_done`.\n\nIf `check` returns `consumer_fenced`, this process no longer owns its Dispatch:\nthe Attempt was re-attached to another worker or settled without you. Stop, do\nnot send `worker_done`, and do not retry the check. An empty `check` never means\nyou were replaced; `consumer_fenced` is the only way you learn that.\n\n## Escalation\n\nEscalate only before completion and only when the coordinator must intervene:\n\n```text\nORCA orchestration send --from --dispatch-capability --type escalation --subject \"Blocked: \" --body \"
\" --task-id --dispatch-id \n```\n\n## Completion\n\nSend exactly one terminal report. `--body` is three sentences: what changed,\nwhat was found, and what remains. Use `--outcome failed` when the requested work\nis not complete; never hide failure in prose or silently exit.\n\nAppend `--files-modified` or `--report-path` only when applicable, using actual\npaths. Do not send documentation placeholders as metadata.\n\n```text\nORCA orchestration send --from --dispatch-capability --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded\n```\n\nAfter `worker_done`, end the dispatched turn and idle. Do not poll, close your\nown terminal, or begin unrelated work. A later direct user instruction is new\nuser-owned work and must not reuse settled lifecycle IDs; a supervised follow-up\narrives with a fresh preamble and Task block.\n" +const ORCHESTRATION_FULL_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nA `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting\nwith `check --wait`. Absence never earns an argv; settlement and pending work still do.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n\n---\n\n# Bundled references\n\nThese references belong to the version-matched guide above. Read only the documents named by its action gates.\n\n\n\n# Coordinator loop\n\nLoad this reference for expanded DAG waves, per-invocation launch preferences,\nsame-terminal reuse, or review ownership. The compact guide remains the source\nof truth for the loop order and completion boundary.\n\n## Ready waves\n\nCreate independent Tasks before the first wait. Encode only real dependencies,\nthen use the ready view as external memory:\n\n```text\nORCA orchestration task-create --spec \"\" --deps --json\nORCA orchestration task-list --ready --brief --json\n```\n\n`--brief` collapses whitespace and caps echoed specs at 160 characters;\n`spec_truncated` identifies shortened rows. Omit it when full specs are needed or\nwhen an older CLI rejects the flag. A nested worker must respect\n`nested_worker_depth_exceeded`; creating another Run does not reset depth.\n\n## Launch preferences\n\nFor a fresh Claude, Codex, or Cursor terminal, `--model` accepts an opaque\nprovider model ID. Pass it only when the user named a model; otherwise omit it\nso the worker inherits the user's configured agent default. Add `--effort` only\nwhen that model supports it:\n\n```text\nORCA orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`; neither option combines with `--terminal`. A\nconnected worker server must advertise launch-preference support before Orca\nforwards either field. Compare `launch.requested` with `launch.effective`; never\nclaim a model or effort from requested arguments alone.\n\n## Reuse after settlement\n\nChoose the terminal's next owner before acknowledging the Delivery. When the\nsame exact agent has immediate follow-up work, recover the proven handle and\ntransfer cleanup ownership to the new Dispatch:\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-start --task --terminal --json\n```\n\nOtherwise explicitly retain or release the settled worker. Do not leave it live\nonly to inspect output; archived output remains available through `worker-read`.\n\n## Review ownership\n\nA review-only `worker_done` authorizes synthesis of findings, not coordinator\nfile edits. Dispatch or hand off fixes unless the user explicitly assigned them\nto the coordinator. If the user's plan names a next owner, post-review fixes and\nPR preparation remain with that owner; the coordinator routes and synthesizes.\n\n\n\n# Legacy contract migration\n\nLoad this reference only for an authority label, adopted Run, compatibility or\nrecovery receipt, or explicit legacy takeover. A newly created attempt always\nuses the current grammar.\n\n## Authority labels\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported\n command printed with the message, using the same selected executable and\n arguments supplied by the original prompt.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded,\n at-least-once cutover replay. Process it idempotently and acknowledge only\n through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or\n lifecycle mutation.\n- An unlabeled current message uses the current guide and grammar.\n\nAn explicitly selected current Run, attested current binding, current Dispatch,\nor federated attachment takes precedence over legacy fallback. A retained\nadoption record alone does not grant mutation authority. If liveness, principal\nownership, capability, or the exact legacy contract is unproven, degrade to\nread-only inspection and never fall back to local execution.\n\nAdoption preserves the live agent process, PTY/session, terminal handle,\ntab/pane, worktree or folder workspace, Task, and Dispatch. It never restarts or\nreplaces the worker and never revives the retired scheduler. Loss of lifecycle\nauthority does not invalidate the existing process, assignment, or filesystem\nwork. Exact recovery may restore the same PTY once in its original inactive\nbackground tab; it must not spawn, write, signal, stop, switch, focus, split, or\ninject a terminal.\n\n## Compatibility recovery\n\nWhen a compatibility response returns structured next-step arguments, execute\nthose exact arguments with the same selected CLI executable. Do not translate\nfrom memory, broaden the recipient, or retry as a current mutation unless the\nreceipt explicitly authorizes it.\n\nA pending ask, reply, final Dispatch settlement, and consuming check have\ndurable recovery identities. Heartbeat and escalation remain at-least-once\nacross a manual contract-boundary retry. If an ask may already have been\nanswered, run the exact non-consuming recovery check printed by Orca before\ncreating any new question. Never guess among identical question threads.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The\ninitial command commits the question, prints its exact\n`ask --resume ` command, and exits with launcher status `75`. Run\nthat exact resume after the launcher or update boundary. For an attested WSL\nlaunch, preserve the printed `orca-ide` executable and distro route. Older WSL\nworkers without launch proof remain lifecycle read-only even while their\nterminal and filesystem work continue.\n\n## Read-only inspection and takeover\n\nRead-only inspection does not consume mail:\n\n```text\nORCA orchestration run-list --json\nORCA orchestration run-show --id run_legacy_local --json\nORCA orchestration run-show --id --json\nORCA orchestration task-list --run --json\nORCA orchestration inbox --full --json\nORCA orchestration check --terminal --peek --format --json\nORCA terminal read --terminal --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\n`run_legacy_local` is an empty audit tombstone after adoption. Find the ordinary\nRun whose objective is `Recovered orchestration work from a contract update`.\n\nOnly when the original coordinator is unavailable or cannot prove retained\nauthority may a new live coordinator take over from its own terminal:\n\n```text\nORCA orchestration run-use --id --takeover-legacy --json\nORCA orchestration check --run --json\n```\n\nTakeover binds the authenticated invoking terminal; `--from` cannot nominate\nanother coordinator. It fences only the old coordinator and moves pending mail\ninto current Run delivery. It preserves live workers, Tasks, Dispatches, processes, and files.\nNever take over while the original coordinator is actively coordinating.\n\nDo not launch a replacement editor merely because Orca updated or authority is\nunclear. Keep the original worker as the only editor until a stable handoff\npoint, then use a fresh current Dispatch in a conflict-free placement.\n\n\n\n# Low-level topology\n\nLoad this reference only when `worker-start` cannot express required custom argv\nor terminal topology. It is not the normal supervised loop and is never a full\nhandoff recipe.\n\n```text\nORCA terminal create --worktree active --title --command \"\" --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\nORCA orchestration dispatch --task --to --inject --json\n```\n\nWait for readiness only when startup could lose injected input. Prefer\nagent-first `worker-start` whenever its argv and topology are sufficient.\n\n`dispatch --inject` creates authoritative Task/Dispatch context but deliberately\nkeeps an operator-created process unsupervised: it creates no supervised worker\nresource row. `worker-show`, `worker-read`, and `worker-list` report the lane as\n`unsupervised`; `worker-stop` and `worker-abandon` do not close that process, and\nsettled retain/release take no process action.\n\nUse `worker-start --terminal ` when lifecycle ownership of an existing\nagent terminal is required. Never imply that low-level dispatch retroactively\nowns a process, never use it to route around the nested-depth limit, and never\nuse it for an ownership handoff.\n\n\n\n# Messaging and gates\n\nLoad this reference for inbox replay, attempt-specific guidance, group\naddresses, blocking questions, or coordinator-managed DAG decisions.\n\nA successful `send` proves durable enqueue. Wake and nudge are best-effort\nattention only: neither proves the recipient read the message, began a turn, or\naccepted steering.\n\n## Coordinator delivery loop\n\n`check` names its caller with `--terminal ` and is the only verb that\nrejects `--from`. Omit `--terminal` inside an Orca terminal, where Orca resolves\nthe caller; pass it explicitly from anywhere else, including a dispatched\nworker reading coordinator follow-ups.\n\nA consuming coordinator `check` returns the bound Run's oldest FIFO Delivery,\nup to 50 messages, and replays that exact batch until acknowledged. Process\nevery row and required terminal ownership decision before `--ack`. Type filters\ndecide when a waiter wakes; they do not authorize skipping older actionable\nmail. A Delivery therefore always carries the whole FIFO batch whatever its\ntypes, and a `check` without `--wait` hands that batch over unfiltered.\n`--peek` and `--all` are read-only inspection, not progress through the\ncoordinator inbox.\n\nAn empty wait or timeout is a checkpoint. Continue rolling waits until every\nexpected Dispatch settles. Heartbeat or visible activity means alive, not done.\n\n## Addresses\n\nUse a stable Dispatch address for attempt-specific coordinator guidance:\n\n```text\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nDo not substitute a remote terminal handle. Omit `--from` for ordinary\ncoordinator calls; a dispatched worker instead copies the exact `--from` and\ncapability arguments in its preamble. `check` is the exception: it identifies\nits caller with `--terminal`, never `--from`.\n\nGroup addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`,\n`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Use them only for\nintentional fan-out status or questions. `worker_done`, heartbeat, and other\nDispatch lifecycle messages never target groups.\n\n## Questions and gates\n\nA worker uses `ask`; its timeout leaves one durable question pending, which the\nworker resumes by message ID. The coordinator answers that message with `reply`.\n\nUse a gate only for a coordinator-owned Task-DAG decision:\n\n```text\nORCA orchestration gate-create --task --question \"\" --options --json\nORCA orchestration gate-resolve --id --resolution \"\" --json\nORCA orchestration gate-list --task --json\n```\n\nPass `json_array` using the quoting rules of the active shell; do not copy POSIX\nsingle-quote syntax into PowerShell or `cmd.exe`.\n\nDo not create a gate merely to answer a worker's `ask`.\n\n\n\n# Placement and remote execution\n\nLoad this reference before creating a new worktree or placing work through SSH,\nWSL, or another connected Orca server.\n\n## Placement choices\n\nA fresh worker means a fresh agent terminal, not a new Git worktree. Use the\ncurrent or an exact existing workspace by default. Create a worktree only when\nthe user requested one or a concrete checkout or filesystem conflict makes\nsharing unsafe.\n\n```text\n# Current workspace; setup is not rerun.\nORCA orchestration worker-start --task --worktree current --agent codex --json\n\n# Stacked child worktree.\nORCA orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n\n# Independent top-level worktree.\nORCA orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nCurrent and exact existing workspaces create a fresh terminal unless\n`--terminal` is explicit. Folder workspaces are first-class; do not invoke Git\nor require worktree lineage when the selected workspace is a folder.\n\nRegister a folder workspace through project setup. `repo add --path `\nrequires a valid Git repository and rejects a plain directory:\n\n```text\nORCA project setup-existing-folder --project --host --path --kind folder --json\n```\n\nThen place work on the returned workspace with an exact selector. A worktree\nselector needs the full `::` value Orca returned, passed as\n`id:`; a bare repo id is not a worktree id. `new-child` and\n`new-top-level` are worktree creation and do not apply to a folder.\n\nNew worktrees use agent-first creation and run setup by default. Preserve the\nrepository's startup policy: `start-immediately` can report setup as `running`,\nwhile `wait-for-setup` gates prompt delivery on success. Orca lineage, Git base,\nfilesystem isolation, coordination parentage, UI grouping, and execution host\nare separate decisions.\n\n## Connected servers\n\nThe Run and Tasks remain authoritative on the current server. `--on` selects\nonly the worker's execution server and appears only on `worker-start`:\n\n```text\nORCA orchestration worker-start --task --on --worktree new-top-level --repo --name --agent codex --setup run --json\n```\n\nRemote `current` and `new-child` are invalid because they are ambiguous across\nservers. Use an exact discovered remote workspace, or `new-top-level` with an\nexact remote repository selector. After start, route every follow-up, read,\nstop, and cleanup by Dispatch ID; never repeat `--on` or substitute a remote\nterminal handle.\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\nORCA orchestration worker-list --run --include-remote --json\n```\n\n`worker-list` reads local fleet state only; enumerate remote workers with\n`--include-remote` or every one of them reads `unverifiable`. Scope every list\nwith `--run `: unscoped, it reports every Dispatch this runtime has\nrecorded, and the workers you are waiting on are lost in that history.\n\n## Execution-host and mixed-version floor\n\nThe execution host owns process, filesystem, transcript, stop, and cleanup\nfacts. Render only `live`, `unverifiable`, or `exited`. Connection loss, relay\nabsence, missing client inventory, or timeout yields `unverifiable`, never\nsynthetic exit and never a client-local substitute action.\n\nClients and servers update independently. Optional response fields may be\nabsent. Forward model/effort, transcript reads, cleanup, or another new remote\noperation only when the peer advertises the relevant capability; unknown stream\nopcodes can be silently dropped. A narrow unsupported response may degrade to a\ndocumented older path, but must not broaden the target or cross the execution\nboundary. Changing host-published content reaches old clients even without a\nwire-shape change, so preserve established semantics or negotiate the behavior.\n\nFor WSL, use the exact executable and arguments returned by Orca so the distro\nand packaged launcher remain bound. Do not translate a printed `orca-ide`\nrecovery command into a PATH-resolved local command.\n\n\n\n# Recovery and cleanup\n\nLoad this reference only after a failed/stopped/unknown attempt, explicit retry\ndecision, stop/abandon request, retention request, or uncertain release.\n\n| Proven state | Safe action |\n| ----------------------- | ------------------------------------------------------------------ |\n| `ready` or active | Keep waiting; optionally read bounded output |\n| `failed` or `stopped` | Start a replacement with `--retry-of`; repeat placement explicitly |\n| `outcome_unknown` | Inspect, then choose `worker-stop` or explicit `worker-abandon` |\n| Accepted `worker_done` | Reuse, retain, or release |\n| Remote contact lost | Preserve `unverifiable`; do not stop or retry from absence alone |\n| `unverifiable` liveness | Keep waiting or inspect; never stop, abandon, retry, or release |\n| Proven `exited` agent | Enumerate with `worker-list`; follow its `nextAction` |\n\n## Inspect before acting\n\n```text\nORCA orchestration worker-list --run --json\nORCA orchestration worker-list --run --include-remote --json\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\n```\n\n`worker-list` is the enumerating command and the authority on agent liveness:\neach row carries `projection.liveness`, `projection.attention.categories`,\n`projection.attention.requiresAction`, and a literal `projection.nextAction`\nargv to run. Always scope it with `--run `; an unscoped list reports\nevery Dispatch this runtime has ever recorded and buries the live ones.\n`worker-show`'s `observation.status` is PTY liveness only, so a `live` terminal\nwhose agent died at a trust prompt still reads `live` there.\n\nWhen the two disagree, the fleet verdict decides — unless the fleet row is\n`unverifiable` for a reason that names a gap on this client rather than a fact\nabout the worker. `missing_status`, `host_unavailable`, and\n`capability_unsupported` are such gaps: the first means this runtime holds no\nstatus row, the second that it could not ask the execution host at all, and the\nthird that a stale peer answered but lacks the fleet-snapshot capability.\nAgainst any of them, a `worker-show` verdict sourced from the execution host is\nthe better evidence and outranks the row. Only `host_unavailable` is contact\nloss; the other two mean the host was never asked or answered without the\ncapability.\n\nThis never promotes absence. `unverifiable` from either command still authorizes\nnothing — only a positive `live` or `exited` verdict does.\n\nA worker started with `--on ` reads `unverifiable` until you\nenumerate with `--include-remote`, which asks its execution host for the\nverdict. Past 100 rows the response pages, so follow `page.nextCursor` with\n`--cursor ` until `page.hasMore` is false.\n\n## Stall needs positive evidence\n\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Only then choose `worker-stop` or `worker-abandon`.\n\n`unverifiable` is always absence — `missing_status`, `stale_status`,\n`restored_unconfirmed`, or a remote worker with no connection — and a null\n`agentWait` or an unchanged `worker-read` tail is that same absence seen again.\nAbsence never authorizes stop, abandon, retry, or release: keep waiting, or\ninspect until you hold one of the positive signals above. A `nextAction` that\nnames an inspecting command is asking for evidence, not for cleanup.\n\n`worker-read --source auto` uses a proven provider transcript when available and\notherwise returns bounded terminal output with a typed `fallbackReason`.\nContinue with its top-level cursor, which is pinned to that source. If Orca\nreports `source_changed`, restart without the old cursor. A bounded initial\ntranscript tail can return an EOF cursor that follows only newly appended records;\nread `contentComplete`, `clipping`, and `warnings` before assuming omitted older\nrecords are pageable. Never guess a provider session ID, transcript path, or\nremote terminal handle.\n\n## Was the mutation applied?\n\nWhen a mutation's response was lost and named no Dispatch, do not replay blind.\nEvery orchestration mutation accepts `--retry-request `, which reuses one\noperation identity so Orca can replay, join, or recover it instead of starting a\nduplicate. Ask what happened first:\n\n```text\nORCA orchestration request-show --request --json\n```\n\n`completed` means the mutation already took effect; read its recorded receipt\ninstead of rerunning. `pending` means the original mutation is still running or\nOrca restarted before recording its outcome; replay the original command with\n`--retry-request `. `absent` means this runtime holds no receipt\nunder your caller identity — that is not proof nothing happened, so inspect the\naffected Task, Dispatch, and terminal before deciding whether to retry.\n\nWhen a worker's terminal accepted input but the submit is unconfirmed, use\n`terminal send --wait-submit `: it observes the accepted prompt for that\nlong and, on timeout, returns the input-accepted receipt without resending.\n\n## Refused starts\n\n`dispatch` and `worker-start` refuse the following preflight cases with a stable\n`error.code`; read it before choosing a recovery, and treat `error.data.nextSteps`\nas the exact recovery text. Older hosts may omit `data`, so treat every field as\noptional.\n\n| Code | Meaning | Recovery |\n| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |\n| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |\n| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |\n| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |\n\n## Retry, stop, and abandon\n\nRetry only a positively proven failed or stopped attempt. Name the failed Task\nwith `--task`, since `--spec` creates a new one. Placement is never silently\ninherited:\n\n```text\nORCA orchestration worker-start --task --retry-of --worktree --agent --json\n```\n\nAfter three consecutive failures for one Task, its dispatch context\ncircuit-breaks and the Task is failed. Do not route around that boundary with a\nnew Run or an unrelated Dispatch.\n\nFor `outcome_unknown`, inspect first, then make an explicit choice:\n\n```text\nORCA orchestration worker-stop --dispatch --json\nORCA orchestration worker-abandon --dispatch --json\n```\n\n`worker-stop` closes only the exact proven supervised agent terminal. It never\ndeletes the worktree, setup terminal, configured tabs, or unrelated processes.\n`worker-abandon` fences orchestration while accepting that resources may remain\nlive; it performs no remote, process, or filesystem action.\n\n## Retain and release\n\n```text\nORCA orchestration worker-retain --dispatch --json\nORCA orchestration worker-release --dispatch --json\n```\n\nRetain only when the user explicitly wants the settled terminal kept live.\nRelease works after succeeded and failed reports, archives readable output, and\ncloses only the exact terminal owned by that settled Dispatch. Replays may call\nrelease again safely. Reused, pre-existing, setup, coordinator, active,\nuser-taken-over, and unproven terminals are retained.\n\nA `worker-start` that failed before its agent was ready still owns the terminal\nit created. Its receipt names `worker-release`, and `worker-list` reports that\nrow as `reclaimable`; release it there rather than closing the terminal by hand.\n\nNever release because of timeout, TUI idle, heartbeat, status, question,\nescalation, or stale/rejected completion. If the receipt says `release_pending`\nor `release_unknown`, follow its exact recovery action. Never substitute\n`terminal close`.\n\n`orchestration reset` is destructive recovery. Do not run it during active\ncoordination unless the user explicitly abandons that state.\n\n\n\n# Worker contract\n\nThe injected preamble is authoritative. Copy its command rather than\nreconstructing flags. In particular, preserve the exact executable, worker\nhandle, Dispatch capability, Task ID, and Dispatch ID.\n\n## Heartbeat\n\nSend heartbeats only at the cadence required by the live preamble. Skip them\nwhile blocked inside `ask` or `check --wait`; those calls are liveness signals.\n\n```text\nORCA orchestration send --from --dispatch-capability --type heartbeat --subject \"alive\" --task-id --dispatch-id --phase \"\"\n```\n\nUse typed lifecycle flags, not a hand-written JSON payload. A heartbeat proves\nliveness, never completion.\n\n## Ask and resume\n\nUse Orca `ask` whenever the coordinator must answer. Never open a local question\nTUI the coordinator cannot answer.\n\n```text\nORCA orchestration ask --from --dispatch-capability --question \"\" --options \",\" --timeout-ms 600000\n\nORCA orchestration ask --from --dispatch-capability --resume --timeout-ms 600000\n```\n\nA timeout or disconnect leaves the original question pending. Resume its\nmessage ID; do not create a duplicate question.\n\n## Reading coordinator follow-ups\n\nThe coordinator steers a running worker with `send --to dispatch:`. That\nenqueue is durable but does not interrupt you, so nothing arrives unless you\nlook:\n\n```text\nORCA orchestration check --terminal --json\n```\n\nRun it at each natural checkpoint — before starting a new file, after a test\nrun — and once more immediately before `worker_done`, so a redirect or a\ncancellation lands before the Task settles. `check` names its caller with\n`--terminal`, never `--from`. Stop checking after `worker_done`.\n\nIf `check` returns `consumer_fenced`, this process no longer owns its Dispatch:\nthe Attempt was re-attached to another worker or settled without you. Stop, do\nnot send `worker_done`, and do not retry the check. An empty `check` never means\nyou were replaced; `consumer_fenced` is the only way you learn that.\n\n## Escalation\n\nEscalate only before completion and only when the coordinator must intervene:\n\n```text\nORCA orchestration send --from --dispatch-capability --type escalation --subject \"Blocked: \" --body \"
\" --task-id --dispatch-id \n```\n\n## Completion\n\nSend exactly one terminal report. `--body` is three sentences: what changed,\nwhat was found, and what remains. Use `--outcome failed` when the requested work\nis not complete; never hide failure in prose or silently exit.\n\nAppend `--files-modified` or `--report-path` only when applicable, using actual\npaths. Do not send documentation placeholders as metadata.\n\n```text\nORCA orchestration send --from --dispatch-capability --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded\n```\n\nAfter `worker_done`, end the dispatched turn and idle. Do not poll, close your\nown terminal, or begin unrelated work. A later direct user instruction is new\nuser-owned work and must not reuse settled lifecycle IDs; a supervised follow-up\narrives with a fresh preamble and Task block.\n" // oxfmt-ignore const ORCHESTRATION_COORDINATOR_LOOP_REFERENCE_MARKDOWN = "# Coordinator loop\n\nLoad this reference for expanded DAG waves, per-invocation launch preferences,\nsame-terminal reuse, or review ownership. The compact guide remains the source\nof truth for the loop order and completion boundary.\n\n## Ready waves\n\nCreate independent Tasks before the first wait. Encode only real dependencies,\nthen use the ready view as external memory:\n\n```text\nORCA orchestration task-create --spec \"\" --deps --json\nORCA orchestration task-list --ready --brief --json\n```\n\n`--brief` collapses whitespace and caps echoed specs at 160 characters;\n`spec_truncated` identifies shortened rows. Omit it when full specs are needed or\nwhen an older CLI rejects the flag. A nested worker must respect\n`nested_worker_depth_exceeded`; creating another Run does not reset depth.\n\n## Launch preferences\n\nFor a fresh Claude, Codex, or Cursor terminal, `--model` accepts an opaque\nprovider model ID. Pass it only when the user named a model; otherwise omit it\nso the worker inherits the user's configured agent default. Add `--effort` only\nwhen that model supports it:\n\n```text\nORCA orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`; neither option combines with `--terminal`. A\nconnected worker server must advertise launch-preference support before Orca\nforwards either field. Compare `launch.requested` with `launch.effective`; never\nclaim a model or effort from requested arguments alone.\n\n## Reuse after settlement\n\nChoose the terminal's next owner before acknowledging the Delivery. When the\nsame exact agent has immediate follow-up work, recover the proven handle and\ntransfer cleanup ownership to the new Dispatch:\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-start --task --terminal --json\n```\n\nOtherwise explicitly retain or release the settled worker. Do not leave it live\nonly to inspect output; archived output remains available through `worker-read`.\n\n## Review ownership\n\nA review-only `worker_done` authorizes synthesis of findings, not coordinator\nfile edits. Dispatch or hand off fixes unless the user explicitly assigned them\nto the coordinator. If the user's plan names a next owner, post-review fixes and\nPR preparation remain with that owner; the coordinator routes and synthesizes.\n" diff --git a/src/cli/handlers/orchestration-worker-cli.test.ts b/src/cli/handlers/orchestration-worker-cli.test.ts index cddd36a4cf7..931c4477855 100644 --- a/src/cli/handlers/orchestration-worker-cli.test.ts +++ b/src/cli/handlers/orchestration-worker-cli.test.ts @@ -104,6 +104,34 @@ describe('orchestration worker-start CLI contract', () => { expect(process.exitCode).toBeUndefined() }) + it.each(['succeeded', 'failed'])( + 'accepts a successful start whose task already %s', + async (workerOutcome) => { + const receipt = { + taskId: 'task_1', + dispatchId: 'ctx_1', + state: 'ready', + stage: 'settled', + workerOutcome, + effects: [], + residualResources: [] + } + callMock.mockResolvedValue({ result: receipt }) + await invokeWorkerStart( + new Map([ + ['task', 'task_1'], + ['from', 'term_coord'] + ]) + ) + expect(process.exitCode).toBeUndefined() + expect(printResult).toHaveBeenCalledWith( + expect.objectContaining({ result: receipt }), + true, + expect.any(Function) + ) + } + ) + it('capability-gates and forwards per-invocation launch preferences', async () => { callMock .mockResolvedValueOnce({ diff --git a/src/cli/runtime/websocket-transport.test.ts b/src/cli/runtime/websocket-transport.test.ts index 162b431f1bf..4c178b71fd8 100644 --- a/src/cli/runtime/websocket-transport.test.ts +++ b/src/cli/runtime/websocket-transport.test.ts @@ -18,6 +18,7 @@ import { launchOrcaApp } from './launch' import { addEnvironmentFromPairingCode } from './environments' import { RuntimeClientError } from './types' import { + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY, MIN_COMPATIBLE_RUNTIME_CLIENT_VERSION, @@ -70,6 +71,7 @@ describe('CLI remote WebSocket transport', () => { expect(runtime.authFrames).toContainEqual( expect.objectContaining({ clientCapabilities: [ + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY, SESSION_TABS_AUTHORITATIVE_INVENTORY_RUNTIME_CAPABILITY, AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, diff --git a/src/main/agent-hooks/hook-stdin-contract.ts b/src/main/agent-hooks/hook-stdin-contract.ts index acba7927650..de77b2f3e9f 100644 --- a/src/main/agent-hooks/hook-stdin-contract.ts +++ b/src/main/agent-hooks/hook-stdin-contract.ts @@ -74,21 +74,38 @@ export const WINDOWS_HOOK_STDIN_DRAIN_LABEL = 'orca_agent_hook_drain_stdin' export const WINDOWS_HOOK_STDIN_READER = '"%SystemRoot%\\System32\\more.com"' export const WINDOWS_HOOK_STDIN_DRAIN_COMMAND = `${WINDOWS_HOOK_STDIN_READER} >nul 2>nul` +// The Orca context a hook needs before it may own stdin; see the rule below. +const WINDOWS_HOOK_ENVIRONMENT_VARS = [ + 'ORCA_AGENT_HOOK_PORT', + 'ORCA_AGENT_HOOK_TOKEN', + 'ORCA_PANE_KEY' +] as const + // Why (#11549): missing Orca context means the hook ran outside an Orca pane, where the caller // may abandon stdin rather than close it — a read-to-EOF then blocks forever and strands a // visible window per hook event. The Windows rule: a hook must check the Orca env before it // owns stdin, and exit without reading when the env is missing — the payload is discarded on -// that path anyway. This applies to .cmd, the copilot .ps1, and the Git Bash kimi .sh alike. +// that path anyway. This applies to .cmd, the copilot .ps1, and the Git Bash kimi .sh alike, +// and to the launchers that own stdin themselves when the managed script is missing. // POSIX hooks keep capture-first: their callers close stdin, and exiting mid-write there // surfaces as EPIPE the agent can see (#8110). export function buildWindowsHookEnvironmentGuardLines(): string[] { - return [ - 'if "%ORCA_AGENT_HOOK_PORT%"=="" exit /b 0', - 'if "%ORCA_AGENT_HOOK_TOKEN%"=="" exit /b 0', - 'if "%ORCA_PANE_KEY%"=="" exit /b 0' - ] + return WINDOWS_HOOK_ENVIRONMENT_VARS.map((name) => `if "%${name}%"=="" exit /b 0`) } +/** The same guard in sh, for the Git Bash hooks and launchers that run on Windows. + * Default-formed because a static hook precheck (Grok) rejects a bare reference it + * cannot resolve. POSIX hosts keep capture-first — this is the Windows rule only. */ +export const WINDOWS_GIT_BASH_HOOK_ENVIRONMENT_GUARD = `if ${WINDOWS_HOOK_ENVIRONMENT_VARS.map( + (name) => `[ -z "\${${name}-}" ]` +).join(' || ')}; then exit 0; fi` + +/** The same guard for a PowerShell hook or launcher. Anything that reaches + * `[Console]::In.ReadToEnd()` must run this first, or it inherits #11549. */ +export const WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD = `if (${WINDOWS_HOOK_ENVIRONMENT_VARS.map( + (name) => `-not $env:${name}` +).join(' -or ')}) { exit 0 }` + export function buildWindowsHookStdinDrainEpilogue(): string[] { return [`:${WINDOWS_HOOK_STDIN_DRAIN_LABEL}`, WINDOWS_HOOK_STDIN_DRAIN_COMMAND, 'exit /b 0'] } diff --git a/src/main/agent-hooks/installer-utils.test.ts b/src/main/agent-hooks/installer-utils.test.ts index 215979206e9..cbe29ee1ca1 100644 --- a/src/main/agent-hooks/installer-utils.test.ts +++ b/src/main/agent-hooks/installer-utils.test.ts @@ -31,7 +31,10 @@ import { type HooksConfig } from './installer-utils' import { buildPosixAgentHookPostCommand } from './hook-post-command' -import { POSIX_HOOK_STDIN_DRAIN_COMMAND } from './hook-stdin-contract' +import { + POSIX_HOOK_STDIN_DRAIN_COMMAND, + WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD +} from './hook-stdin-contract' import { wrapRuntimeHomeHookCommand } from './runtime-home-hook-command' let tmpDir: string @@ -618,7 +621,10 @@ function expectedDecodedWindowsHookCommand(scriptPath: string): string { // Why: the execution-policy bypass rides in the payload, not on the command // line, so the launcher cannot spell the AV-blocked flag triple (#16003). // Why: PowerShell progress CLIXML corrupts consumers that merge stderr into JSON stdout. - return `$ProgressPreference='SilentlyContinue'; try { Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force -ErrorAction SilentlyContinue } catch {}; if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; [Console]::In.ReadToEnd() | Out-Null; exit 0` + // Why the guard is spelled by import: the launcher owns stdin on the missing-script path, + // so it obeys the shared Windows rule (#11549), and re-typing it here would let the two + // drift back apart. + return `$ProgressPreference='SilentlyContinue'; try { Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force -ErrorAction SilentlyContinue } catch {}; if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; ${WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD}; [Console]::In.ReadToEnd() | Out-Null; exit 0` } describe('wrapWindowsHookCommand', () => { @@ -640,15 +646,23 @@ describe('wrapWindowsHookCommand', () => { ) }) - it('emits fallback stdout when the managed script is missing', () => { - const command = wrapWindowsHookCommand( - 'C:\\hooks\\cursor-hook.cmd', - {}, - { fallbackStdout: '{"permission":"allow"}' } - ) - expect(decodeWindowsHookCommand(command)).toContain( - 'Write-Output \'{"permission":"allow"}\'; exit 0' + // Why the ordering matters: a gate event reads silence as deny (#2426), and outside an + // Orca pane the guard exits before the read — so an answer placed after the drain never + // reaches the agent at all when the caller abandons the pipe (#11549). + it('answers before it guards, and guards before it owns stdin', () => { + const decoded = decodeWindowsHookCommand( + wrapWindowsHookCommand( + 'C:\\hooks\\cursor-hook.cmd', + {}, + { fallbackStdout: '{"permission":"allow"}' } + ) ) + const answer = decoded.indexOf('Write-Output \'{"permission":"allow"}\'') + const guard = decoded.indexOf(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD) + const ownsStdin = decoded.indexOf('[Console]::In.ReadToEnd()') + expect(answer).toBeGreaterThan(-1) + expect(guard).toBeGreaterThan(answer) + expect(ownsStdin).toBeGreaterThan(guard) }) // Why: a user profile path like `C:\Users\Jane Doe` is the regression from diff --git a/src/main/agent-hooks/installer-utils.ts b/src/main/agent-hooks/installer-utils.ts index 8667c418492..a53721d42fe 100644 --- a/src/main/agent-hooks/installer-utils.ts +++ b/src/main/agent-hooks/installer-utils.ts @@ -16,6 +16,7 @@ import { grantDirAcl, isPermissionError } from '../win32-utils' import { resolveHooksJsonWritePath } from './hook-config-write-path' import { writeRollingFileBackup } from '../rolling-file-backup' import { wrapWindowsPowerShellEncodedCommand } from './windows-powershell-hook-launcher' +import { WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD } from './hook-stdin-contract' export type HookCommandConfig = { type: 'command' @@ -131,7 +132,10 @@ export function wrapWindowsHookCommand( options.fallbackStdout === undefined ? '' : `Write-Output ${quotePowerShellString(options.fallbackStdout)}; ` - const command = `${envPrefix}if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; [Console]::In.ReadToEnd() | Out-Null; ${fallback}exit 0` + // Why the order: answer first (a gate event reads silence as deny), then the shared + // env guard, and only then own stdin — outside an Orca pane the caller may abandon the + // pipe, and ReadToEnd would strand the launcher there forever (#11549). + const command = `${envPrefix}if (Test-Path -LiteralPath ${quoted} -PathType Leaf) { & ${quoted}; exit $LASTEXITCODE }; ${fallback}${WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD}; [Console]::In.ReadToEnd() | Out-Null; exit 0` return wrapWindowsPowerShellEncodedCommand(command) } diff --git a/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts b/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts index af70f6f54f0..dea4545ad11 100644 --- a/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts +++ b/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts @@ -63,12 +63,26 @@ import { KimiHookService } from '../kimi/hook-service' import { openClaudeHookService } from '../openclaude/hook-service' import { wrapPosixHookCommand, wrapWindowsHookCommand } from './installer-utils' -import { POSIX_HOOK_STDIN_READER } from './hook-stdin-contract' +import { + POSIX_HOOK_STDIN_READER, + WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD +} from './hook-stdin-contract' import { wrapRuntimeHomeHookCommand } from './runtime-home-hook-command' import { createAgentHookMemorySftp } from './agent-hook-memory-sftp.test-fixture' import { findGitBash } from './windows-git-bash-path.test-fixture' +/** The launchers ship their command base64'd; assert the shape they actually run. */ +function decodeEncodedPowerShellCommand(command: string): string { + const encoded = command.match(/-EncodedCommand\s+(\S+)/) + expect(encoded, 'launcher carries an encoded command').not.toBeNull() + return Buffer.from(encoded![1], 'base64').toString('utf16le') +} + const REMOTE_HOME = '/home/dev' +// Why all three: Windows reports a write to a pipe whose reader is gone as any of these, +// depending on whether the read handle, the pipe, or the process went first. Enumerating +// them keeps the guard-exit legs from failing on which race the host happened to run. +const WRITER_BROKEN_BY_EARLY_EXIT = ['EPIPE', 'ECONNRESET', 'EOF'] const LARGE_PAYLOAD = Buffer.alloc(1_000_000, 'x') // Why: a developer box may set HKCU\...\Command Processor\AutoRun, which cmd.exe runs before any @@ -156,7 +170,10 @@ type HookRun = { function runHookProcess( executable: string, args: string[], - env: NodeJS.ProcessEnv + env: NodeJS.ProcessEnv, + // Why: `abandon` leaves the pipe open and unwritten — the shape a caller outside an Orca + // pane produces, and the only one that can catch a read-to-EOF that never returns (#11549). + stdin: 'close' | 'abandon' = 'close' ): Promise { return new Promise((resolve, reject) => { const child = spawn(executable, args, { env, stdio: ['pipe', 'pipe', 'pipe'] }) @@ -164,8 +181,9 @@ function runHookProcess( let stderr = '' let stdout = '' const timeout = setTimeout(() => { + child.stdin.destroy() child.kill('SIGKILL') - reject(new Error('hook did not finish after stdin closed')) + reject(new Error(`hook did not finish with stdin ${stdin}d`)) }, 10_000) child.on('error', (error) => { clearTimeout(timeout) @@ -182,7 +200,9 @@ function runHookProcess( clearTimeout(timeout) resolve({ exitCode, stdinErrors, stderr, stdout }) }) - child.stdin.end(LARGE_PAYLOAD) + if (stdin === 'close') { + child.stdin.end(LARGE_PAYLOAD) + } }) } @@ -303,6 +323,31 @@ describe('Windows managed hook stdin structure', () => { expect(copilot.indexOf('if (-not $env:ORCA_AGENT_HOOK_PORT')).toBeLessThan( copilot.indexOf('[Console]::In.ReadToEnd()') ) + // Why: the two encoded-PowerShell launchers own stdin themselves when the managed + // script is missing, so the same guard has to precede their ReadToEnd — and the + // fallback answer has to precede the guard, or a gate event outside a pane is + // answered with silence, which reads as deny (#2426/#15462). + for (const [name, command] of [ + [ + 'wrapWindowsHookCommand', + wrapWindowsHookCommand('C:\\missing\\orca-hook.cmd', {}, { fallbackStdout: '{}' }) + ], + [ + 'wrapRuntimeHomeHookCommand', + wrapRuntimeHomeHookCommand('missing-orca-hook', { neutralJsonWhenMissing: true }) + ] + ] as const) { + const decoded = decodeEncodedPowerShellCommand(command) + expect(decoded, `${name} decoded`).toContain(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD) + expect(decoded.indexOf("Write-Output '{}'"), `${name} answers first`).toBeLessThan( + decoded.indexOf(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD) + ) + expect( + decoded.indexOf(WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD), + `${name} guards before owning stdin` + ).toBeLessThan(decoded.indexOf('[Console]::In.ReadToEnd()')) + } + const kimi = readFileSync(join(hooksDir, 'kimi-hook.sh'), 'utf8') expect(kimi.indexOf('if [ -z "$ORCA_AGENT_HOOK_PORT" ]')).toBeGreaterThan(-1) expect(kimi.indexOf('if [ -z "$ORCA_AGENT_HOOK_PORT" ]')).toBeLessThan( @@ -365,12 +410,11 @@ describe('Windows managed hook stdin structure', () => { const result = await runHookProcess(executable, args, hookEnvironment()) expect(result.exitCode, `${fileName} exit code`).toBe(0) // Why (#11549 class): every Windows-local hook exits before owning stdin when the - // Orca env is missing, so the writer may break — EPIPE, or ECONNRESET when Windows - // tears the pipe down first. hookEnvironment() strips every ORCA_* var, so this - // relaxation only ever covers the missing-env path — a happy-path case added to - // this loop must not reuse it. + // Orca env is missing, so the writer may break. hookEnvironment() strips every + // ORCA_* var, so this relaxation only ever covers the missing-env path — a + // happy-path case added to this loop must not reuse it. for (const error of result.stdinErrors) { - expect(['EPIPE', 'ECONNRESET'], `${fileName} stdin error`).toContain(error.code) + expect(WRITER_BROKEN_BY_EARLY_EXIT, `${fileName} stdin error`).toContain(error.code) } } @@ -395,9 +439,42 @@ describe('Windows managed hook stdin structure', () => { } ] for (const launcher of launcherCases) { - const result = await runHookProcess(launcher.executable, launcher.args, hookEnvironment()) - expect(result.exitCode, `${launcher.name} exit code`).toBe(0) - expect(result.stdinErrors, `${launcher.name} stdin errors`).toHaveLength(0) + // Why (#11549 class): a launcher that reaches an interpreter owns stdin for a + // missing script exactly like a managed script does, so it obeys the same rule — + // drain inside a pane, exit before reading outside one. Its writer may therefore + // break on the missing-env leg, and must not on the in-pane leg. + const outside = await runHookProcess( + launcher.executable, + launcher.args, + hookEnvironment() + ) + expect(outside.exitCode, `${launcher.name} exit code`).toBe(0) + for (const error of outside.stdinErrors) { + expect(WRITER_BROKEN_BY_EARLY_EXIT, `${launcher.name} stdin error`).toContain( + error.code + ) + } + const insideAPane = await runHookProcess( + launcher.executable, + launcher.args, + hookEnvironment({ + ORCA_AGENT_HOOK_PORT: '59999', + ORCA_AGENT_HOOK_TOKEN: 'token', + ORCA_PANE_KEY: 'tab:leaf' + }) + ) + expect(insideAPane.exitCode, `${launcher.name} in-pane exit code`).toBe(0) + expect(insideAPane.stdinErrors, `${launcher.name} in-pane stdin errors`).toHaveLength(0) + // Why this leg and not a shape assertion: an unguarded ReadToEnd exits fine when + // the writer closes the pipe. Only a caller that abandons it strands the launcher, + // which is what left a console per hook event on the reporting hosts. + const abandoned = await runHookProcess( + launcher.executable, + launcher.args, + hookEnvironment(), + 'abandon' + ) + expect(abandoned.exitCode, `${launcher.name} abandoned-stdin exit code`).toBe(0) } } finally { homedirMock.mockImplementation(() => process.env.HOME ?? tmpdir()) diff --git a/src/main/agent-hooks/runtime-home-hook-command.ts b/src/main/agent-hooks/runtime-home-hook-command.ts index 3a6f0d20725..e56fc603b43 100644 --- a/src/main/agent-hooks/runtime-home-hook-command.ts +++ b/src/main/agent-hooks/runtime-home-hook-command.ts @@ -1,4 +1,8 @@ -import { POSIX_HOOK_STDIN_DRAIN_COMMAND } from './hook-stdin-contract' +import { + POSIX_HOOK_STDIN_DRAIN_COMMAND, + WINDOWS_GIT_BASH_HOOK_ENVIRONMENT_GUARD, + WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD +} from './hook-stdin-contract' import { encodeWindowsPowerShellHookCommand, WINDOWS_POWERSHELL_HOOK_SWITCHES @@ -19,16 +23,30 @@ export function wrapRuntimeHomeHookCommand( const windowsScript = `"\${HOME-}/.orca/agent-hooks/${scriptBaseName}.cmd"` const posixScript = `"\${HOME-}/.orca/agent-hooks/${scriptBaseName}.sh"` const drain = POSIX_HOOK_STDIN_DRAIN_COMMAND - const missingScriptFallback = options.neutralJsonWhenMissing ? `${drain}; printf '{}\\n'` : drain + const neutralJson = options.neutralJsonWhenMissing ? `printf '{}\\n'` : '' + // Why two forms: the missing-script fallback owns stdin, so it follows the rule of the host + // it lands on. POSIX callers close the pipe, so capture-first is safe there and a mid-write + // exit stays visible as EPIPE (#8110). A Windows caller may abandon the pipe, so there the + // answer comes first and the drain only runs with an Orca env behind it (#11549). + const posixMissingScriptFallback = neutralJson ? `${drain}; ${neutralJson}` : drain + const windowsMissingScriptFallback = [ + ...(neutralJson ? [neutralJson] : []), + WINDOWS_GIT_BASH_HOOK_ENVIRONMENT_GUARD, + drain + ].join('; ') + // Why platform-selected even when HOME is unset: which stdin rule applies follows the + // caller, not the reason the script could not be found. + const missingScriptFallback = `case "\${OSTYPE-}" in msys*|cygwin*|win32*) ${windowsMissingScriptFallback} ;; *) ${posixMissingScriptFallback} ;; esac` const powershell = '"${SYSTEMROOT-}/System32/WindowsPowerShell/v1.0/powershell.exe"' const powershellFallback = options.neutralJsonWhenMissing ? "; Write-Output '{}'" : '' - const powershellCommand = `$homePath = $env:HOME -replace '^/([A-Za-z])/', '$1:/'; $scriptPath = Join-Path $homePath '.orca\\agent-hooks\\${scriptBaseName}.cmd'; if (Test-Path -LiteralPath $scriptPath -PathType Leaf) { & $scriptPath; exit $LASTEXITCODE }; [Console]::In.ReadToEnd() | Out-Null${powershellFallback}; exit 0` + // Why the order: answer first, then the shared env guard, then own stdin — see wrapWindowsHookCommand. + const powershellCommand = `$homePath = $env:HOME -replace '^/([A-Za-z])/', '$1:/'; $scriptPath = Join-Path $homePath '.orca\\agent-hooks\\${scriptBaseName}.cmd'; if (Test-Path -LiteralPath $scriptPath -PathType Leaf) { & $scriptPath; exit $LASTEXITCODE }${powershellFallback}; ${WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD}; [Console]::In.ReadToEnd() | Out-Null; exit 0` const encodedCommand = encodeWindowsPowerShellHookCommand(powershellCommand) // Why: the Git Bash and native Windows launchers must spell the same switches — window suppression (#14815) and an AV verdict on the shape (#16003) both hit either path. const powershellInvocation = `${powershell} ${WINDOWS_POWERSHELL_HOOK_SWITCHES} -EncodedCommand ${encodedCommand}` - const encodedWindowsBranch = `if [ -f ${powershell} ]; then ${powershellInvocation}; else ${missingScriptFallback}; fi` - const windowsBranch = `if [ -f ${windowsScript} ]; then case "\${HOME-}" in ${WINDOWS_GIT_BASH_RUNTIME_HOME_UNSAFE}) ${encodedWindowsBranch} ;; *) ${windowsScript} ;; esac; else ${missingScriptFallback}; fi` - const posixBranch = `if [ -f ${posixScript} ] && [ -r ${posixScript} ] && [ -x ${posixScript} ]; then /bin/sh ${posixScript}; else ${missingScriptFallback}; fi` + const encodedWindowsBranch = `if [ -f ${powershell} ]; then ${powershellInvocation}; else ${windowsMissingScriptFallback}; fi` + const windowsBranch = `if [ -f ${windowsScript} ]; then case "\${HOME-}" in ${WINDOWS_GIT_BASH_RUNTIME_HOME_UNSAFE}) ${encodedWindowsBranch} ;; *) ${windowsScript} ;; esac; else ${windowsMissingScriptFallback}; fi` + const posixBranch = `if [ -f ${posixScript} ] && [ -r ${posixScript} ] && [ -x ${posixScript} ]; then /bin/sh ${posixScript}; else ${posixMissingScriptFallback}; fi` // Why: OSTYPE is shell-owned, so platform selection adds no process to every hook invocation. return `if [ -z "\${HOME-}" ]; then ${missingScriptFallback}; else case "\${OSTYPE-}" in msys*|cygwin*|win32*) ${windowsBranch} ;; *) ${posixBranch} ;; esac; fi` } diff --git a/src/main/ai-vault/session-scanner-jsonl-reader.test.ts b/src/main/ai-vault/session-scanner-jsonl-reader.test.ts new file mode 100644 index 00000000000..386510ffd89 --- /dev/null +++ b/src/main/ai-vault/session-scanner-jsonl-reader.test.ts @@ -0,0 +1,104 @@ +import { expect, it, vi } from 'vitest' +import { consumeCompleteJsonlLines } from './session-scanner-jsonl-reader' + +const source = vi.hoisted(() => ({ chunks: [] as Buffer[] })) +vi.mock('../native-chat/wsl-transcript-fs-access', () => ({ + openTranscriptReadStream: async function* () { + yield* source.chunks + } +})) + +it('copies only the carried line when the next chunk contains many complete lines', async () => { + source.chunks = Array.from({ length: 100 }, () => Buffer.from(`${'a\n'.repeat(1000)}x`)) + const original = Buffer.concat + let copied = 0 + const concat = vi.spyOn(Buffer, 'concat').mockImplementation((chunks, total) => { + copied += total ?? chunks.reduce((sum, chunk) => sum + chunk.length, 0) + return original(chunks, total) + }) + let lines = 0 + let result: Awaited> + try { + result = await consumeCompleteJsonlLines({ + path: '/log', + start: 0, + onLine: () => { + lines += 1 + } + }) + } finally { + concat.mockRestore() + } + expect(lines).toBe(100000) + expect(result!).toEqual({ consumedThrough: 200099, trailingPartialLine: 'x', bytesRead: 200100 }) + expect(copied).toBeLessThan(1000) +}) + +it('preserves UTF-8/CRLF carry, byte callbacks and stop offsets', async () => { + source.chunks = [Buffer.from('ab\r'), Buffer.from('\ncd\npartial')] + const lines: string[] = [] + expect( + await consumeCompleteJsonlLines({ + path: '/log', + start: 5, + onLine: () => {}, + onLineBytes: (line) => lines.push(line.toString()) + }) + ).toEqual({ consumedThrough: 12, trailingPartialLine: 'partial', bytesRead: 14 }) + expect(lines).toEqual(['ab', 'cd']) + let stopped = false + expect( + await consumeCompleteJsonlLines({ + path: '/log', + start: 5, + onLine: () => { + stopped = true + }, + shouldStop: () => stopped + }) + ).toEqual({ consumedThrough: 9, trailingPartialLine: null, bytesRead: 14 }) + const unicode = Buffer.from('🦀\n') + source.chunks = [unicode.subarray(0, 2), unicode.subarray(2)] + const onLine = vi.fn() + await consumeCompleteJsonlLines({ path: '/log', start: 0, onLine }) + expect(onLine).toHaveBeenCalledWith('🦀') +}) + +// Why: a chunk boundary is not aligned to anything — it can land mid-record, +// mid-UTF-8-sequence, between CR and LF, or on an empty line. A dropped or +// merged line here silently corrupts an agent transcript, and a wrong +// `consumedThrough` makes the next incremental scan resume mid-line. +it('yields identical lines and resume offsets for every single-byte chunk split', async () => { + const bigRecord = `{"d":${'"'.padEnd(2000, 'z')}"}` + const expectedLines = [ + '{"a":1}', // plain LF record + '{"b":"🦀 é 𝄞"}', // CRLF record whose content is 2/3/4-byte UTF-8 + '', // empty line + '', // empty CRLF line + '{"c":"x\ry"}', // lone CR inside a record + bigRecord // single record larger than any carried prefix + ] + const trailing = '{"partial":' // final line with no trailing newline + const buffer = Buffer.from( + `{"a":1}\n{"b":"🦀 é 𝄞"}\r\n\n\r\n{"c":"x\ry"}\n${bigRecord}\n${trailing}`, + 'utf-8' + ) + const expectedConsumed = buffer.length - Buffer.byteLength(trailing) + + for (let cut = 0; cut <= buffer.length; cut++) { + source.chunks = [buffer.subarray(0, cut), buffer.subarray(cut)].filter((c) => c.length > 0) + const lines: string[] = [] + const result = await consumeCompleteJsonlLines({ + path: '/log', + start: 41, + onLine: (line) => lines.push(line) + }) + expect({ cut, lines, ...result }).toEqual({ + cut, + lines: expectedLines, + consumedThrough: 41 + expectedConsumed, + trailingPartialLine: trailing, + bytesRead: buffer.length + }) + } +}) diff --git a/src/main/ai-vault/session-scanner-jsonl-reader.ts b/src/main/ai-vault/session-scanner-jsonl-reader.ts index 613c50ef0ba..6734339dc2f 100644 --- a/src/main/ai-vault/session-scanner-jsonl-reader.ts +++ b/src/main/ai-vault/session-scanner-jsonl-reader.ts @@ -36,23 +36,24 @@ export async function consumeCompleteJsonlLines(args: { remainderLength += chunk.length continue } - const data = - remainderLength > 0 - ? Buffer.concat([...remainderParts, chunk], remainderLength + chunk.length) - : chunk - remainderParts = [] - remainderLength = 0 + const data = chunk + const carriedLength = remainderLength let lineStart = 0 let newlineIndex = data.indexOf(NEWLINE_BYTE, lineStart) while (newlineIndex !== -1) { - let lineEnd = newlineIndex - if (lineEnd > lineStart && data[lineEnd - 1] === CARRIAGE_RETURN_BYTE) { - lineEnd-- + let line = data.subarray(lineStart, newlineIndex) + // Only the first line of a chunk can carry a prefix; resetting inside the + // branch keeps the common per-line path allocation-free. + if (remainderLength > 0) { + line = Buffer.concat([...remainderParts, line], remainderLength + line.length) + remainderParts = [] + remainderLength = 0 } + const lineEnd = line.at(-1) === CARRIAGE_RETURN_BYTE ? line.length - 1 : line.length if (args.onLineBytes) { - args.onLineBytes(data.subarray(lineStart, lineEnd)) + args.onLineBytes(line.subarray(0, lineEnd)) } else { - args.onLine(data.toString('utf-8', lineStart, lineEnd)) + args.onLine(line.toString('utf-8', 0, lineEnd)) } lineStart = newlineIndex + 1 if (args.shouldStop?.()) { @@ -61,7 +62,7 @@ export async function consumeCompleteJsonlLines(args: { } newlineIndex = data.indexOf(NEWLINE_BYTE, lineStart) } - consumedThrough += lineStart + consumedThrough += carriedLength + lineStart if (stopped) { remainderParts = [] remainderLength = 0 diff --git a/src/main/antigravity/hook-script.ts b/src/main/antigravity/hook-script.ts index 67f1f639ef9..fb27ad63494 100644 --- a/src/main/antigravity/hook-script.ts +++ b/src/main/antigravity/hook-script.ts @@ -88,7 +88,10 @@ export function getManagedScript(target: 'local' | 'posix' = 'local'): string { export function getWindowsWrapperScript(eventName: string): string { return [ '@echo off', - 'setlocal', + // Why (#9358/#9941): `!` is legal in the hooks path, and inherited delayed expansion + // eats it out of the percent-expanded `%~dp0` — the wrapper then misses the core and + // silently falls back on every event. Same reason the core disables it. + 'setlocal DisableDelayedExpansion', `set "ORCA_ANTIGRAVITY_EVENT=${eventName}"`, 'set "ORCA_ANTIGRAVITY_CORE=%~dp0antigravity-hook.cmd"', 'if exist "%ORCA_ANTIGRAVITY_CORE%" (', @@ -102,8 +105,8 @@ export function getWindowsWrapperScript(eventName: string): string { ') else (', ' echo {}', ')', - // Why: when the shared core script is missing, this wrapper becomes the - // stdin owner and must finish the agent's payload write before returning. + // Missing-core fallbacks obey the same outside-Orca stdin guard as the core. + ...buildWindowsHookEnvironmentGuardLines(), WINDOWS_HOOK_STDIN_DRAIN_COMMAND, 'exit /b 0', '' diff --git a/src/main/antigravity/windows-hook-payload-delivery.test.ts b/src/main/antigravity/windows-hook-payload-delivery.test.ts index 8261cc3ce91..6c9ca08bd27 100644 --- a/src/main/antigravity/windows-hook-payload-delivery.test.ts +++ b/src/main/antigravity/windows-hook-payload-delivery.test.ts @@ -28,7 +28,8 @@ vi.mock('os', async (importOriginal) => { import { AntigravityHookService } from './hook-service' import { ANTIGRAVITY_EVENTS, ANTIGRAVITY_PRE_TOOL_USE_DECISION } from './hook-events' -import { getManagedScript } from './hook-script' +import { getManagedScript, getWindowsWrapperScript } from './hook-script' +import { WINDOWS_HOOK_STDIN_DRAIN_COMMAND } from '../agent-hooks/hook-stdin-contract' // Why (#9358/#9941): `!` is legal in a Windows path and in a pane key. Under inherited // delayed expansion cmd eats it out of a percent-expanded curl argument, so bake one into @@ -91,17 +92,23 @@ async function startHookListener(): Promise<{ type HookRun = { exitCode: number | null; stdout: string; stderr: string; timedOut: boolean } +// Why spell `/v`: `cmd /d /c ` is the chain in the bug report's process trace, +// and it inherits HKCU\...\Command Processor\DelayedExpansion. Naming the state makes the +// hostile half reachable on any host — under `/v:on` cmd eats `!` out of every percent +// expansion (#9358/#9941), and a harness pinned to `/v:off` could never fail on it. +type DelayedExpansion = 'on' | 'off' +const DELAYED_EXPANSION_STATES = ['off', 'on'] as const satisfies readonly DelayedExpansion[] + function runWrapper( wrapperPath: string, env: NodeJS.ProcessEnv, // Why: `null` abandons stdin instead of closing it — the shape a caller outside an Orca // pane produces, and the only way to prove the env guard exits before reading (#11549). - stdinPayload: string | null = PAYLOAD + stdinPayload: string | null = PAYLOAD, + delayedExpansion: DelayedExpansion = 'off' ): Promise { return new Promise((resolve, reject) => { - // Why: mirror how Antigravity spawns the hook — `cmd /c `, the exact - // chain in the bug report's process trace. - const child = spawn('cmd.exe', ['/d', '/c', wrapperPath], { + const child = spawn('cmd.exe', [`/v:${delayedExpansion}`, '/d', '/c', wrapperPath], { stdio: ['pipe', 'pipe', 'pipe'], windowsHide: true, env @@ -111,6 +118,7 @@ function runWrapper( let timedOut = false const timer = setTimeout(() => { timedOut = true + child.stdin.destroy() child.kill('SIGKILL') }, 15_000) child.on('error', (error) => { @@ -154,6 +162,24 @@ function expectedStdout(eventName: string): string { // Why: runs on every platform — the live delivery suite below is Windows-only, so this // keeps a POSIX-only CI leg from letting the interpreter back into the hot path. describe('Antigravity Windows hook post command', () => { + it.each(ANTIGRAVITY_EVENTS)('guards missing-core stdin for $eventName', ({ eventName }) => { + const script = getWindowsWrapperScript(eventName) + const drain = script.indexOf(WINDOWS_HOOK_STDIN_DRAIN_COMMAND) + const answer = script.lastIndexOf('echo {}') + expect(drain).toBeGreaterThan(answer) + for (const key of ['ORCA_AGENT_HOOK_PORT', 'ORCA_AGENT_HOOK_TOKEN', 'ORCA_PANE_KEY']) { + const guard = script.indexOf(`if "%${key}%"=="" exit /b 0`) + expect(guard, key).toBeGreaterThan(answer) + expect(guard, key).toBeLessThan(drain) + } + }) + + // Why (#9358/#9941): `%~dp0` carries the hooks path, so an inherited delayed expansion eats + // a `!` out of it and the wrapper silently misses the core on every event. + it.each(ANTIGRAVITY_EVENTS)('disables delayed expansion for $eventName', ({ eventName }) => { + expect(getWindowsWrapperScript(eventName)).toContain('setlocal DisableDelayedExpansion') + }) + it('posts through curl.exe rather than a PowerShell interpreter', () => { vi.spyOn(process, 'platform', 'get').mockReturnValue('win32') const script = getManagedScript('local') @@ -185,7 +211,10 @@ describe.skipIf(process.platform !== 'win32')('Antigravity Windows hook payload }) it('delivers every event wrapper payload to the listener without spawning PowerShell', async () => { - home = mkdtempSync(join(tmpdir(), 'orca-antigravity-hook-')) + // Why the `!` in the directory: it lands in the wrapper's `%~dp0`, which is what an + // inherited delayed expansion eats (#9358/#9941). Without it the `/v:on` leg below + // proves nothing about the core lookup. + home = mkdtempSync(join(tmpdir(), 'orca-antigravity-hook!bang-')) homedirMock.mockReturnValue(home) expect(new AntigravityHookService().install().state).toBe('installed') @@ -204,34 +233,42 @@ describe.skipIf(process.platform !== 'win32')('Antigravity Windows hook payload ORCA_WORKTREE_ID: WORKTREE_ID }) - for (const event of ANTIGRAVITY_EVENTS) { - const label = event.eventName - const before = listener.posts.length - const result = await runWrapper(join(hooksDir, event.windowsWrapperFileName), env) + for (const delayedExpansion of DELAYED_EXPANSION_STATES) { + for (const event of ANTIGRAVITY_EVENTS) { + const label = `${event.eventName} (/v:${delayedExpansion})` + const before = listener.posts.length + const result = await runWrapper( + join(hooksDir, event.windowsWrapperFileName), + env, + PAYLOAD, + delayedExpansion + ) - expect(result.timedOut, `${label} timed out`).toBe(false) - expect(result.exitCode, `${label} exit code`).toBe(0) - expect(result.stderr, `${label} stderr`).toBe('') - // Why: Antigravity reads silence on PreToolUse as deny (#2426), so the gate answer - // must survive the transport change. - expect(result.stdout.trim(), `${label} stdout`).toBe(expectedStdout(label)) + expect(result.timedOut, `${label} timed out`).toBe(false) + expect(result.exitCode, `${label} exit code`).toBe(0) + expect(result.stderr, `${label} stderr`).toBe('') + // Why: Antigravity reads silence on PreToolUse as deny (#2426), so the gate answer + // must survive the transport change. + expect(result.stdout.trim(), `${label} stdout`).toBe(expectedStdout(event.eventName)) - const posts = listener.posts.slice(before) - expect(posts, `${label} posted exactly one hook`).toHaveLength(1) - // Why: byte-exact, not "non-empty" — PowerShell recoded this body through the console - // code page, and a silently corrupted payload still looks posted. - expect(posts[0].payload, `${label} payload`).toBe(PAYLOAD) - expect(posts[0].hookEventName, `${label} hook_event_name`).toBe(label) - // Why: the `!` in both values is the delayed-expansion regression guard. - expect(posts[0].paneKey, `${label} paneKey`).toBe(PANE_KEY) - expect(posts[0].worktreeId, `${label} worktreeId`).toBe(WORKTREE_ID) - expect(posts[0].token, `${label} token`).toBe(HOOK_TOKEN) - expect(posts[0].contentType, `${label} content-type`).toContain( - 'application/x-www-form-urlencoded' - ) + const posts = listener.posts.slice(before) + expect(posts, `${label} posted exactly one hook`).toHaveLength(1) + // Why: byte-exact, not "non-empty" — PowerShell recoded this body through the console + // code page, and a silently corrupted payload still looks posted. + expect(posts[0].payload, `${label} payload`).toBe(PAYLOAD) + expect(posts[0].hookEventName, `${label} hook_event_name`).toBe(event.eventName) + // Why: the `!` in both values is the delayed-expansion regression guard — it is the + // `/v:on` leg that can actually fail on it. + expect(posts[0].paneKey, `${label} paneKey`).toBe(PANE_KEY) + expect(posts[0].worktreeId, `${label} worktreeId`).toBe(WORKTREE_ID) + expect(posts[0].token, `${label} token`).toBe(HOOK_TOKEN) + expect(posts[0].contentType, `${label} content-type`).toContain( + 'application/x-www-form-urlencoded' + ) + } } - // Why: five wrapper launches plus a real install can overrun the default under load. - }, 60_000) + // Why: ten wrapper launches plus a real install can overrun the default under load. + }, 90_000) // Why (#15117): Antigravity fires some events with no stdin at all. PowerShell substituted // `{}` before posting; curl forwards the empty body, so prove the post still happens — the @@ -264,6 +301,67 @@ describe.skipIf(process.platform !== 'win32')('Antigravity Windows hook payload expect(listener.posts[0].hookEventName).toBe('PreInvocation') }, 30_000) + // Why a helper: the missing-core cases all need a real install with the core removed, which + // is the shape an AV quarantine or a half-finished uninstall leaves behind. + async function installWithoutCore(): Promise { + home = mkdtempSync(join(tmpdir(), 'orca-antigravity-fallback-')) + homedirMock.mockReturnValue(home) + expect(new AntigravityHookService().install().state).toBe('installed') + const hooksDir = join(home, '.orca', 'agent-hooks') + rmSync(join(hooksDir, 'antigravity-hook.cmd')) + return hooksDir + } + + it.each(['ORCA_AGENT_HOOK_PORT', 'ORCA_AGENT_HOOK_TOKEN', 'ORCA_PANE_KEY'])( + 'answers every missing-core event with abandoned stdin and no %s', + async (missingKey) => { + const hooksDir = await installWithoutCore() + const listener = await startHookListener() + server = listener.server + const env = hookEnvironment({ + USERPROFILE: home, + HOME: home, + ORCA_AGENT_HOOK_PORT: String(listener.port), + ORCA_AGENT_HOOK_TOKEN: HOOK_TOKEN, + ORCA_PANE_KEY: PANE_KEY, + [missingKey]: '' + }) + for (const event of ANTIGRAVITY_EVENTS) { + const result = await runWrapper(join(hooksDir, event.windowsWrapperFileName), env, null) + expect(result.timedOut, event.eventName).toBe(false) + expect(result.exitCode, event.eventName).toBe(0) + expect(result.stdout.trim(), event.eventName).toBe(expectedStdout(event.eventName)) + expect(result.stderr, event.eventName).toBe('') + } + expect(listener.posts).toHaveLength(0) + }, + 90_000 + ) + + // Why: the guard must not cost the valid path its drain — with the Orca env present the + // fallback still owns stdin, so the agent's payload write completes instead of breaking. + it('still drains a closed payload for every missing-core event inside a pane', async () => { + const hooksDir = await installWithoutCore() + const listener = await startHookListener() + server = listener.server + const env = hookEnvironment({ + USERPROFILE: home, + HOME: home, + ORCA_AGENT_HOOK_PORT: String(listener.port), + ORCA_AGENT_HOOK_TOKEN: HOOK_TOKEN, + ORCA_PANE_KEY: PANE_KEY + }) + for (const event of ANTIGRAVITY_EVENTS) { + const result = await runWrapper(join(hooksDir, event.windowsWrapperFileName), env) + expect(result.timedOut, event.eventName).toBe(false) + expect(result.exitCode, event.eventName).toBe(0) + expect(result.stdout.trim(), event.eventName).toBe(expectedStdout(event.eventName)) + expect(result.stderr, event.eventName).toBe('') + } + // Why: the fallback answers the agent but has no core to post through. + expect(listener.posts).toHaveLength(0) + }, 60_000) + it('exits without reading stdin when the pane env is missing', async () => { home = mkdtempSync(join(tmpdir(), 'orca-antigravity-hook-')) homedirMock.mockReturnValue(home) diff --git a/src/main/automations/external-job-mappers.ts b/src/main/automations/external-job-mappers.ts index d3648af446a..60677dda99c 100644 --- a/src/main/automations/external-job-mappers.ts +++ b/src/main/automations/external-job-mappers.ts @@ -71,7 +71,7 @@ function mapExternalRuns({ .map((run, index) => { const runAt = asString(run.run_at) ?? asString(run.runAt) const id = asString(run.id) ?? `${jobId}:${runAt ?? index}` - return { + const mapped: ExternalAutomationRun = { id, managerId, provider, @@ -83,15 +83,15 @@ function mapExternalRuns({ error: asString(run.error), outputPath: asString(run.output_path) ?? asString(run.outputPath) } + return { run: mapped, time: runAt ? Date.parse(runAt) : Number.NaN } }) .sort((a, b) => { - const aTime = a.runAt ? Date.parse(a.runAt) : Number.NaN - const bTime = b.runAt ? Date.parse(b.runAt) : Number.NaN - if (Number.isFinite(aTime) && Number.isFinite(bTime)) { - return bTime - aTime + if (Number.isFinite(a.time) && Number.isFinite(b.time)) { + return b.time - a.time } - return b.id.localeCompare(a.id) + return b.run.id.localeCompare(a.run.id) }) + .map(({ run }) => run) } function hermesScheduleDisplay(job: ExternalJobRecord): string { diff --git a/src/main/automations/external-job-run-sorting.test.ts b/src/main/automations/external-job-run-sorting.test.ts new file mode 100644 index 00000000000..f477fc4afcb --- /dev/null +++ b/src/main/automations/external-job-run-sorting.test.ts @@ -0,0 +1,38 @@ +import { expect, it, vi } from 'vitest' +import { mapHermesJobs, mapOpenClawJobs } from './external-job-mappers' + +it.each([mapHermesJobs, mapOpenClawJobs])( + 'parses run dates once and preserves provider fallback ordering', + (mapJobs) => { + const runs = Array.from({ length: 2000 }, (_, i) => ({ + id: String(i), + run_at: + i % 137 === 0 + ? 'invalid' + : new Date(1700000000000 + ((i * 173) % 1999) * 1000).toISOString(), + output_content: `Output ${i}`, + status: 'completed' + })) + const parse = vi.spyOn(Date, 'parse') + let expected: typeof runs + let jobs: ReturnType + try { + expected = [...runs].sort((a, b) => { + const left = Date.parse(a.run_at), + right = Date.parse(b.run_at) + return Number.isFinite(left) && Number.isFinite(right) + ? right - left + : b.id.localeCompare(a.id) + }) + expect(parse.mock.calls.length).toBeGreaterThan(10_000) + parse.mockClear() + jobs = mapJobs('manager', [{ id: 'job', runs }]) + expect(parse).toHaveBeenCalledTimes(2000) + } finally { + parse.mockRestore() + } + expect(jobs[0].runs.map((run) => run.id)).toEqual(expected.map((run) => run.id)) + expect(jobs[0].runs.every((run) => run.outputContent === `Output ${run.id}`)).toBe(true) + expect(jobs[0].runs.every((run) => !('time' in run))).toBe(true) + } +) diff --git a/src/main/browser/remote-browser-socks-buffering.test.ts b/src/main/browser/remote-browser-socks-buffering.test.ts new file mode 100644 index 00000000000..badae864399 --- /dev/null +++ b/src/main/browser/remote-browser-socks-buffering.test.ts @@ -0,0 +1,117 @@ +import { EventEmitter } from 'node:events' +import { createServer, type Socket } from 'node:net' +import { PassThrough } from 'node:stream' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { RemoteBrowserSocksServer } from './remote-browser-socks-server' + +vi.mock('node:net', () => ({ + createServer: vi.fn(() => ({ listening: false })) +})) + +function setup(requestTail: Buffer = Buffer.alloc(0)) { + const upstream = new PassThrough() + const write = vi.spyOn(upstream, 'write') + const opened = Promise.withResolvers() + const open = vi.fn(() => opened.promise) + const server = new RemoteBrowserSocksServer({ open }) + const socket = Object.assign(new EventEmitter(), { + remoteAddress: '127.0.0.1', + destroyed: false, + write: vi.fn(() => true), + pause: vi.fn(), + end: vi.fn((_reply, callback) => callback()), + pipe: vi.fn(), + destroy: vi.fn(() => { + socket.destroyed = true + socket.emit('close') + }) + }) + const accept = vi.mocked(createServer).mock.calls.at(-1)![0] as (socket: Socket) => void + accept(socket as unknown as Socket) + socket.emit('data', Buffer.from([5, 1, 0])) + socket.emit('data', Buffer.concat([Buffer.from([5, 1, 0, 1, 127, 0, 0, 1, 1, 187]), requestTail])) + return { server, socket, upstream, write, opened, open } +} + +afterEach(() => vi.restoreAllMocks()) + +describe('pending browser SOCKS route buffering', () => { + it('copies fragmented pending bytes linearly and forwards every byte at the existing cap', async () => { + const { server, socket, upstream, write, opened } = setup() + const payload = Buffer.alloc(256 * 1024) + for (let index = 0; index < payload.length; index += 1) { + payload[index] = index % 251 + } + let copiedBytes = 0 + const originalCopy = Buffer.prototype.copy + const copy = vi.spyOn(Buffer.prototype, 'copy').mockImplementation(function (target, ...args) { + const copied = originalCopy.call(this, target, ...args) + copiedBytes += copied + return copied + }) + const concat = vi.spyOn(Buffer, 'concat') + try { + for (let index = 0; index < payload.length; index += 256) { + socket.emit('data', payload.subarray(index, index + 256)) + } + expect(concat.mock.calls.length).toBe(0) + expect(copiedBytes).toBeLessThan(payload.length * 3) + opened.resolve(upstream) + await vi.waitFor(() => expect(write).toHaveBeenCalledTimes(1)) + expect(write.mock.calls[0][0]).toEqual(payload) + } finally { + copy.mockRestore() + concat.mockRestore() + await server.close() + upstream.destroy() + } + }) + + it('keeps request-tail bytes ahead of later fragments in the pending payload', async () => { + const tail = Buffer.from('GET / HTTP/1.1\r\n') + const { server, socket, upstream, write, opened } = setup(tail) + const rest = Buffer.from('Host: example.com\r\n\r\n') + try { + for (const byte of rest) { + socket.emit('data', Buffer.from([byte])) + } + opened.resolve(upstream) + await vi.waitFor(() => expect(write).toHaveBeenCalledTimes(1)) + expect(write.mock.calls[0][0]).toEqual(Buffer.concat([tail, rest])) + } finally { + await server.close() + upstream.destroy() + } + }) + + it('rejects one byte beyond the cap and destroys a late upstream without forwarding', async () => { + const { server, socket, upstream, write, opened, open } = setup() + try { + await vi.waitFor(() => expect(open).toHaveBeenCalledTimes(1)) + socket.emit('data', Buffer.alloc(256 * 1024)) + expect(socket.destroyed).toBe(false) + socket.emit('data', Buffer.from([1])) + expect(socket.destroyed).toBe(true) + expect(socket.end.mock.calls[0][0][1]).toBe(1) + opened.resolve(upstream) + await vi.waitFor(() => expect(upstream.destroyed).toBe(true)) + expect(write).not.toHaveBeenCalled() + } finally { + await server.close() + } + }) + + it('discards pending input on client close while the route is opening', async () => { + const { server, socket, upstream, write, opened, open } = setup() + try { + await vi.waitFor(() => expect(open).toHaveBeenCalledTimes(1)) + socket.emit('data', Buffer.from('pending request')) + socket.destroy() + opened.resolve(upstream) + await vi.waitFor(() => expect(upstream.destroyed).toBe(true)) + expect(write).not.toHaveBeenCalled() + } finally { + await server.close() + } + }) +}) diff --git a/src/main/browser/remote-browser-socks-server.ts b/src/main/browser/remote-browser-socks-server.ts index 976ed48af8c..1b72963cf97 100644 --- a/src/main/browser/remote-browser-socks-server.ts +++ b/src/main/browser/remote-browser-socks-server.ts @@ -1,5 +1,7 @@ import { createServer, type Server, type Socket } from 'node:net' import type { Duplex } from 'node:stream' +import { GrowingByteBuffer } from '../../shared/growing-byte-buffer' +import { pipeUpstreamToClient } from './remote-browser-socks-upstream' const SOCKS_VERSION = 5 const SOCKS_NO_AUTH = 0 @@ -106,10 +108,13 @@ export class RemoteBrowserSocksServer { this.clients.add(socket) let phase: 'greeting' | 'request' | 'opening' | 'connected' | 'closed' = 'greeting' let buffered = Buffer.alloc(0) + const pendingUpstream = new GrowingByteBuffer() const timeout = setTimeout(() => socket.destroy(), HANDSHAKE_TIMEOUT_MS) const cleanup = (): void => { phase = 'closed' clearTimeout(timeout) + buffered = Buffer.alloc(0) + pendingUpstream.clear() this.clients.delete(socket) } const finishFailure = (reply: Uint8Array): void => { @@ -119,6 +124,7 @@ export class RemoteBrowserSocksServer { phase = 'closed' clearTimeout(timeout) buffered = Buffer.alloc(0) + pendingUpstream.clear() socket.pause() socket.end(reply, () => socket.destroy()) } @@ -127,13 +133,15 @@ export class RemoteBrowserSocksServer { if (phase === 'closed' || phase === 'connected') { return } - buffered = Buffer.concat([buffered, chunk]) if (phase === 'opening') { - if (buffered.byteLength > MAX_PENDING_UPSTREAM_BYTES) { + if (pendingUpstream.byteLength + chunk.byteLength > MAX_PENDING_UPSTREAM_BYTES) { fail(1) + } else { + pendingUpstream.append(chunk) } return } + buffered = Buffer.concat([buffered, chunk]) if (phase === 'greeting' && buffered.byteLength > MAX_HANDSHAKE_BYTES) { fail(1) return @@ -181,6 +189,8 @@ export class RemoteBrowserSocksServer { return } phase = 'opening' + pendingUpstream.append(buffered) + buffered = Buffer.alloc(0) void Promise.resolve() .then(() => this.open(normalizeListenerWildcard(parsed.target))) .then( @@ -193,9 +203,8 @@ export class RemoteBrowserSocksServer { clearTimeout(timeout) socket.off('data', onData) socket.write(SUCCESS_RESPONSE) - if (buffered.byteLength > 0) { - upstream.write(buffered) - buffered = Buffer.alloc(0) + if (pendingUpstream.byteLength > 0) { + upstream.write(pendingUpstream.takeBuffer()) } socket.pipe(upstream) pipeUpstreamToClient(upstream, socket) @@ -212,22 +221,6 @@ export class RemoteBrowserSocksServer { } } -function pipeUpstreamToClient(upstream: Duplex, socket: Socket): void { - upstream.on('data', (chunk: Buffer) => { - const bytes = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk) - const accepted = socket.write(bytes, (error) => { - if (!error && 'settleRead' in upstream && typeof upstream.settleRead === 'function') { - upstream.settleRead(bytes.byteLength) - } - }) - if (!accepted) { - upstream.pause() - } - }) - socket.on('drain', () => upstream.resume()) - upstream.once('end', () => socket.end()) -} - function parseSocksRequest(buffer: Uint8Array): SocksRequest | null | undefined { if (buffer.byteLength < 4) { return undefined diff --git a/src/main/browser/remote-browser-socks-upstream.ts b/src/main/browser/remote-browser-socks-upstream.ts new file mode 100644 index 00000000000..167563ccad1 --- /dev/null +++ b/src/main/browser/remote-browser-socks-upstream.ts @@ -0,0 +1,18 @@ +import type { Socket } from 'node:net' +import type { Duplex } from 'node:stream' + +export function pipeUpstreamToClient(upstream: Duplex, socket: Socket): void { + upstream.on('data', (chunk: Buffer) => { + const bytes = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk) + const accepted = socket.write(bytes, (error) => { + if (!error && 'settleRead' in upstream && typeof upstream.settleRead === 'function') { + upstream.settleRead(bytes.byteLength) + } + }) + if (!accepted) { + upstream.pause() + } + }) + socket.on('drain', () => upstream.resume()) + upstream.once('end', () => socket.end()) +} diff --git a/src/main/claude-usage/claude-usage-report-aggregation.ts b/src/main/claude-usage/claude-usage-report-aggregation.ts index 9231af85cfa..82aef355b1d 100644 --- a/src/main/claude-usage/claude-usage-report-aggregation.ts +++ b/src/main/claude-usage/claude-usage-report-aggregation.ts @@ -1,3 +1,4 @@ +import { highestUsageKey } from '../usage/highest-usage-key' import type { ClaudeUsageBreakdownKind, ClaudeUsageBreakdownRow, @@ -56,9 +57,8 @@ export function buildSummary( } } - const topModel = [...byModel.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null - const topProject = - [...byProject.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null + const topModel = highestUsageKey(byModel) + const topProject = highestUsageKey(byProject) return { scope, diff --git a/src/main/claude-usage/worktree-attribution-scaling.test.ts b/src/main/claude-usage/worktree-attribution-scaling.test.ts new file mode 100644 index 00000000000..eee23973f13 --- /dev/null +++ b/src/main/claude-usage/worktree-attribution-scaling.test.ts @@ -0,0 +1,53 @@ +import { expect, it, vi } from 'vitest' +import { attributeClaudeUsageTurns } from './worktree-attribution' +import type { ClaudeUsageParsedTurn } from './types' + +vi.mock('node:fs/promises', () => ({ realpath: async (path: string) => path })) + +it('resolves repeated nested and unmatched cwd paths once per attribution batch', async () => { + const lookup = new Map( + Array.from({ length: 100 }, (_, index) => [ + `/repo-${String(index).padStart(3, '0')}`, + { + repoId: `repo-${index}`, + worktreeId: `wt-${index}`, + path: `/repo-${index}`, + displayName: `Repo ${index}` + } + ]) + ) + const input: ClaudeUsageParsedTurn[] = Array.from({ length: 1000 }, (_, index) => ({ + sessionId: String(index), + timestamp: '2026-09-07T00:00:00Z', + model: null, + cwd: index % 2 === 0 ? '/repo-099/nested' : '/outside', + gitBranch: null, + inputTokens: 1, + outputTokens: 1, + cacheReadTokens: 0, + cacheWriteTokens: 0, + cacheWrite1hTokens: 0 + })) + const original = String.prototype.startsWith + let comparisons = 0 + const spy = vi.spyOn(String.prototype, 'startsWith').mockImplementation(function ( + this: string, + search: string, + position?: number + ) { + if (search.slice(0, 6) === '/repo-') { + comparisons += 1 + } + return original.call(this, search, position) + }) + let result: Awaited> + try { + result = await attributeClaudeUsageTurns(input, lookup) + } finally { + spy.mockRestore() + } + expect(comparisons).toBeLessThanOrEqual(200) + expect(result![0].worktreeId).toBe('wt-99') + expect(result![1].worktreeId).toBeNull() + expect(result![1].projectKey).toBe('cwd:/outside') +}) diff --git a/src/main/claude-usage/worktree-attribution.ts b/src/main/claude-usage/worktree-attribution.ts index dedb97025b6..cbf14eb6319 100644 --- a/src/main/claude-usage/worktree-attribution.ts +++ b/src/main/claude-usage/worktree-attribution.ts @@ -105,7 +105,7 @@ export async function attributeClaudeUsageTurns( worktreeLookup: Map ): Promise { const attributed: ClaudeUsageAttributedTurn[] = [] - const canonicalCwdByPath = new Map() + const worktreeByCwd = new Map() for (const turn of turns) { const day = localDayFromTimestamp(turn.timestamp) @@ -119,14 +119,12 @@ export async function attributeClaudeUsageTurns( let projectLabel = getDefaultProjectLabel(turn.cwd) if (turn.cwd) { - let canonicalCwd = canonicalCwdByPath.get(turn.cwd) - if (canonicalCwd === undefined) { - // Why: Claude transcripts repeat the same cwd for many consecutive - // turns. Cache realpath work so attribution scales with unique paths. - canonicalCwd = await canonicalizePath(turn.cwd) - canonicalCwdByPath.set(turn.cwd, canonicalCwd) + let worktree = worktreeByCwd.get(turn.cwd) + if (worktree === undefined) { + const canonicalCwd = await canonicalizePath(turn.cwd) + worktree = findContainingWorktree(canonicalCwd, worktreeLookup) + worktreeByCwd.set(turn.cwd, worktree) } - const worktree = findContainingWorktree(canonicalCwd, worktreeLookup) if (worktree) { repoId = worktree.repoId worktreeId = worktree.worktreeId diff --git a/src/main/claude/claude-background-task-tracker.ts b/src/main/claude/claude-background-task-tracker.ts index a1504a65dec..de24b9a4fba 100644 --- a/src/main/claude/claude-background-task-tracker.ts +++ b/src/main/claude/claude-background-task-tracker.ts @@ -20,14 +20,22 @@ function record(value: unknown): Record | null { return typeof value === 'object' && value !== null ? (value as Record) : null } -function taskId(message: Record): string | null { - const value = message.task_id - return typeof value === 'string' && value.length > 0 && value.length <= MAX_TASK_ID_LENGTH - ? value - : null +/** The bound every task id shares, wherever it enters. An id the roster stores + * becomes a durable entry key, so a provisional one takes the same bound the + * announced path applies — an over-long id is rejected, never truncated. */ +export function isBoundedClaudeTaskId(value: string): boolean { + return value.length > 0 && value.length <= MAX_TASK_ID_LENGTH } -function taskDescription(value: unknown): string | undefined { +/** The task's canonical, resume-stable id. Shared with the subagent roster so + * both readers of this channel agree on what identifies a task. */ +export function claudeTaskId(message: Record): string | null { + const value = message.task_id + return typeof value === 'string' && isBoundedClaudeTaskId(value) ? value : null +} + +/** A task's human label, collapsed and bounded. */ +export function claudeTaskDescription(value: unknown): string | undefined { if (typeof value !== 'string') { return undefined } @@ -107,7 +115,7 @@ export class ClaudeBackgroundTaskTracker { this.replaceAggregateRoster(message.tasks) return true } - const id = taskId(message) + const id = claudeTaskId(message) if (!id) { return false } @@ -126,13 +134,13 @@ export class ClaudeBackgroundTaskTracker { } const existing = this.tasks.get(id) if ( - (patch.is_backgrounded === true || taskDescription(patch.description)) && + (patch.is_backgrounded === true || claudeTaskDescription(patch.description)) && (!this.aggregateRosterObserved || existing) ) { this.upsert(id, { backgrounded: patch.is_backgrounded === true || existing?.backgrounded === true, kind: existing?.kind ?? 'unknown', - description: taskDescription(patch.description) ?? existing?.description + description: claudeTaskDescription(patch.description) ?? existing?.description }) return true } @@ -152,7 +160,7 @@ export class ClaudeBackgroundTaskTracker { this.upsert(id, { backgrounded: message.is_backgrounded === true || kind === 'workflow' || kind === 'monitor', kind, - description: taskDescription(message.description) + description: claudeTaskDescription(message.description) }) return true } @@ -172,14 +180,14 @@ export class ClaudeBackgroundTaskTracker { if (!task || task.ambient === true) { continue } - const id = taskId(task) + const id = claudeTaskId(task) if (!id) { continue } this.tasks.set(id, { backgrounded: true, kind: classifyClaudeBackgroundTaskKind(task.task_type), - description: taskDescription(task.description) + description: claudeTaskDescription(task.description) }) } } diff --git a/src/main/claude/claude-structured-dispatch-content.test.ts b/src/main/claude/claude-structured-dispatch-content.test.ts new file mode 100644 index 00000000000..02ee6fa721d --- /dev/null +++ b/src/main/claude/claude-structured-dispatch-content.test.ts @@ -0,0 +1,165 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import { + claudeDispatchInvokesSlashCommand, + claudeDispatchMessageContent +} from './claude-structured-dispatch-content' + +const PNG = Buffer.from( + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==', + 'base64' +) + +function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem { + return { kind: 'message', role: 'user', blocks } +} + +const REMOTE_IMAGE = { type: 'image-ref' as const, url: 'https://example.test/a.png' } + +describe('claudeDispatchMessageContent', () => { + it('puts the text block last so a slash command still expands with an attachment', async () => { + const content = await claudeDispatchMessageContent( + // The composer builds text-then-images; Claude only treats a leading `/` as a + // command when the LAST block is text. + userMessage([{ type: 'text', text: '/goal ship the parser' }, REMOTE_IMAGE]) + ) + + expect(content).toEqual([ + { type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } }, + { type: 'text', text: '/goal ship the parser' } + ]) + }) + + it('keeps every image ahead of the text and preserves each side’s order', async () => { + const second = { type: 'image-ref' as const, url: 'https://example.test/b.png' } + + const content = await claudeDispatchMessageContent( + userMessage([{ type: 'text', text: 'look' }, REMOTE_IMAGE, second]) + ) + + expect(content.map((part) => (part as { type: string }).type)).toEqual([ + 'image', + 'image', + 'text' + ]) + expect(content[0]).toEqual({ + type: 'image', + source: { type: 'url', url: 'https://example.test/a.png' } + }) + expect(content[1]).toEqual({ + type: 'image', + source: { type: 'url', url: 'https://example.test/b.png' } + }) + }) + + it('sends text alone unchanged', async () => { + const content = await claudeDispatchMessageContent(userMessage([{ type: 'text', text: 'hi' }])) + + expect(content).toEqual([{ type: 'text', text: 'hi' }]) + }) + + it('sends an image with no text', async () => { + const content = await claudeDispatchMessageContent(userMessage([REMOTE_IMAGE])) + + expect(content).toEqual([ + { type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } } + ]) + }) + + it('rejects a message with no renderable block', async () => { + await expect( + claudeDispatchMessageContent(userMessage([{ type: 'text', text: '' }])) + ).rejects.toThrow('Claude dispatch requires text or an image') + }) + + it('rejects a non-user message', async () => { + await expect( + claudeDispatchMessageContent({ + ...userMessage([{ type: 'text', text: 'hi' }]), + role: 'assistant' + }) + ).rejects.toThrow('Claude dispatch accepts only user messages') + }) + + it('joins several text blocks so a command is not stranded ahead of trailing prose', async () => { + // Appending each block would leave `thanks` trailing, and Claude reads only that block. + const content = await claudeDispatchMessageContent( + userMessage([ + { type: 'text', text: '/goal ship' }, + REMOTE_IMAGE, + { type: 'text', text: 'thanks' } + ]) + ) + + expect(content).toEqual([ + { type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } }, + { type: 'text', text: '/goal ship\nthanks' } + ]) + expect(claudeDispatchInvokesSlashCommand(content)).toBe(true) + }) + + it('puts a locally attached image ahead of the text, the shape the composer sends', async () => { + const dir = await mkdtemp(join(tmpdir(), 'claude-dispatch-content-')) + const path = join(dir, 'shot.png') + await writeFile(path, PNG) + + try { + const content = await claudeDispatchMessageContent( + userMessage([ + { type: 'text', text: '/goal ship' }, + { type: 'image-ref', path } + ]) + ) + + expect(content).toEqual([ + { + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: PNG.toString('base64') } + }, + { type: 'text', text: '/goal ship' } + ]) + } finally { + await rm(dir, { recursive: true, force: true }) + } + }) +}) + +describe('claudeDispatchInvokesSlashCommand', () => { + it('reads the trailing prompt Claude recovers, not any text block', () => { + expect( + claudeDispatchInvokesSlashCommand([ + { type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } }, + { type: 'text', text: '/goal ship' } + ]) + ).toBe(true) + // The pre-fix order: Claude recovers no prompt at all, so no command runs. + expect( + claudeDispatchInvokesSlashCommand([ + { type: 'text', text: '/goal ship' }, + { type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } } + ]) + ).toBe(false) + }) + + it('reads the joined prompt, so a command behind leading prose is not one', async () => { + // Keeping the blocks separate would leave `/goal ship` trailing and falsely claim a command. + const content = await claudeDispatchMessageContent( + userMessage([ + { type: 'text', text: 'take a look' }, + { type: 'text', text: '/goal ship' } + ]) + ) + + expect(content).toEqual([{ type: 'text', text: 'take a look\n/goal ship' }]) + expect(claudeDispatchInvokesSlashCommand(content)).toBe(false) + }) + + it('matches untrimmed, as Claude does, and ignores a promptless turn', () => { + expect(claudeDispatchInvokesSlashCommand([{ type: 'text', text: ' /goal ship' }])).toBe(false) + expect(claudeDispatchInvokesSlashCommand([{ type: 'text', text: 'ship it' }])).toBe(false) + expect(claudeDispatchInvokesSlashCommand([])).toBe(false) + }) +}) diff --git a/src/main/claude/claude-structured-dispatch-content.ts b/src/main/claude/claude-structured-dispatch-content.ts index 71f180bc3ac..7df84091b7c 100644 --- a/src/main/claude/claude-structured-dispatch-content.ts +++ b/src/main/claude/claude-structured-dispatch-content.ts @@ -3,6 +3,7 @@ import { open } from 'node:fs/promises' import { extname } from 'node:path' import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' import type { NativeChatBlock } from '../../shared/native-chat-types' +import { claudeRecord } from './claude-structured-item-translation' const MAX_IMAGE_BYTES = 5 * 1024 * 1024 const MAX_IMAGE_COUNT = 20 @@ -88,27 +89,49 @@ async function imageContent( } } +/** + * Claude encodes a user turn as attachment blocks followed by the typed text, and recovers the + * typed prompt by reading only the trailing text block. Verified against the real CLI over + * stream-json: a body ending in an image has no recoverable prompt, so its `/command` reaches + * the model as prose instead of being expanded. + */ export async function claudeDispatchMessageContent( body: AgentJournalMessageItem ): Promise { if (body.role !== 'user') { throw new Error('Claude dispatch accepts only user messages') } - const content: unknown[] = [] + const images: unknown[] = [] + const texts: string[] = [] const imageBudget: ImageBudget = { count: 0, localBytes: 0 } for (const block of body.blocks as NativeChatBlock[]) { if (block.type === 'text' && block.text.length > 0) { - content.push({ type: 'text', text: block.text }) + texts.push(block.text) } else if (block.type === 'image-ref') { - content.push(await imageContent(block, imageBudget)) + images.push(await imageContent(block, imageBudget)) } } + // Join rather than append each block: only the trailing text is read as the prompt, so several + // text blocks would silently discard every one but the last. + const content = texts.length > 0 ? [...images, { type: 'text', text: texts.join('\n') }] : images if (content.length === 0) { throw new Error('Claude dispatch requires text or an image') } return content } +/** The prompt Claude recovers from a dispatch, or null when the turn carries no prompt. */ +function claudeDispatchPrompt(content: readonly unknown[]): string | null { + const last = claudeRecord(content.at(-1)) + return last?.type === 'text' && typeof last.text === 'string' ? last.text : null +} + +/** Mirrors how Claude decides a turn is a command. Untrimmed on purpose: Claude does not trim + * here either, so leading whitespace really does mean no command runs. */ +export function claudeDispatchInvokesSlashCommand(content: readonly unknown[]): boolean { + return claudeDispatchPrompt(content)?.startsWith('/') === true +} + /** * Keep waiter metadata bounded even when a dispatch contains large base64 images. * The digest is only diagnostic: replay acknowledgement must use provider identity. @@ -117,10 +140,7 @@ export function claudeDispatchContentKey(content: readonly unknown[]): string { const digest = createHash('sha256') const summary = content .map((part) => { - const record = - typeof part === 'object' && part !== null && !Array.isArray(part) - ? (part as Record) - : null + const record = claudeRecord(part) const type = typeof record?.type === 'string' ? record.type : 'unknown' if (type === 'text') { return `text:${typeof record?.text === 'string' ? record.text.length : 0}` @@ -136,10 +156,7 @@ export function claudeDispatchContentKey(content: readonly unknown[]): string { }) .join(',') for (const [index, part] of content.entries()) { - const record = - typeof part === 'object' && part !== null && !Array.isArray(part) - ? (part as Record) - : null + const record = claudeRecord(part) const type = typeof record?.type === 'string' ? record.type : 'unknown' digest.update(`${index}:${type}:`) if (type === 'text' && typeof record?.text === 'string') { diff --git a/src/main/claude/claude-structured-dispatch.test.ts b/src/main/claude/claude-structured-dispatch.test.ts index ad09357df58..933bd77774b 100644 --- a/src/main/claude/claude-structured-dispatch.test.ts +++ b/src/main/claude/claude-structured-dispatch.test.ts @@ -423,6 +423,73 @@ describe('Claude structured dispatch image limits', () => { }) }) + it('accepts a slash command sent with an attachment from its result receipt', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { + clientMessageId: 'client-1', + body: userMessage([ + { type: 'text', text: '/permissions' }, + { type: 'image-ref', url: 'https://example.test/a.png' } + ]) + }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + // The mapper moves the image ahead of the prompt, so Claude runs the command and replies + // with a result receipt instead of a user replay. + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'command-result-uuid' + }) + ).toBe(false) + + await expect(dispatched).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'command-result-uuid' } + }) + // The sent order is the fix: the waiter's verdict alone was already what it is today. + expect(session.connection.send).toHaveBeenCalledWith( + expect.objectContaining({ + message: { + role: 'user', + content: [ + { type: 'image', source: { type: 'url', url: 'https://example.test/a.png' } }, + { type: 'text', text: '/permissions' } + ] + } + }) + ) + }) + + it('does not take a result receipt for leading whitespace Claude never reads as a command', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: ' /permissions' }]) + }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'unrelated-result-uuid' + }) + ).toBe(false) + + await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + }) + it('correlates a later slash-command result by user_message_uuid despite a timed-out slash waiter', async () => { const session = sessionFor() const first = dispatchClaudeTurn( diff --git a/src/main/claude/claude-structured-dispatch.ts b/src/main/claude/claude-structured-dispatch.ts index b7619a1e94e..3080b5479e4 100644 --- a/src/main/claude/claude-structured-dispatch.ts +++ b/src/main/claude/claude-structured-dispatch.ts @@ -12,6 +12,7 @@ import type { ClaudeDispatchWaiter, ClaudeSession } from './claude-structured-se import { readClaudeFrameString } from './claude-structured-init-proof' import { claudeDispatchContentKey, + claudeDispatchInvokesSlashCommand, claudeDispatchMessageContent } from './claude-structured-dispatch-content' @@ -231,9 +232,9 @@ export async function dispatchClaudeTurn( return { state: 'rejected', reason: (error as Error).message } } const dispatchSequence = ++session.dispatchSequence - const acceptsResult = input.body.blocks.some( - (block) => block.type === 'text' && block.text.trimStart().startsWith('/') - ) + // Read the sent content, not the journal blocks: only the mapped trailing prompt decides + // whether Claude runs a command, so the two cannot disagree about which frame settles this. + const acceptsResult = claudeDispatchInvokesSlashCommand(content) const sentUuid = randomUUID() const replay = waitForReplay( session, diff --git a/src/main/claude/claude-structured-item-translation.ts b/src/main/claude/claude-structured-item-translation.ts index c0ce20908d8..1c1673b59cd 100644 --- a/src/main/claude/claude-structured-item-translation.ts +++ b/src/main/claude/claude-structured-item-translation.ts @@ -74,6 +74,18 @@ export function claudeMessageIdentity( return { provider: 'claude', sessionId: envelope.sessionId, uuid: envelope.uuid } } +/** User bubbles belong to the submitted message; SDK user frames carry echoes + * and tool results, so a user envelope keeps only its tool results. */ +export function claudeOutputEnvelope(envelope: ClaudeMessageEnvelope): ClaudeMessageEnvelope { + if (envelope.role !== 'user') { + return envelope + } + return { + ...envelope, + content: envelope.content.filter((part) => claudeRecord(part)?.type === 'tool_result') + } +} + function messageBlocks(envelope: ClaudeMessageEnvelope): NativeChatBlock[] { const blocks: NativeChatBlock[] = [] for (const value of envelope.content) { diff --git a/src/main/claude/claude-structured-journal-translation-subagents.test.ts b/src/main/claude/claude-structured-journal-translation-subagents.test.ts new file mode 100644 index 00000000000..3b280020c31 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation-subagents.test.ts @@ -0,0 +1,259 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { + NativeChatSubagentEntry, + NativeChatSubagentGroupBlock +} from '../../shared/native-chat-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +const GROUP_ITEM_ID = 'claude-subagents:claude-session:user-1' + +/** The union's other arms carry no client message id, so reading one narrows. */ +function orcaClientMessageId(identity: AgentJournalItemIdentity): string | null { + return identity.provider === 'orca' ? identity.clientMessageId : null +} + +function harness() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: vi.fn(), + publish: vi.fn() + } + const translator = createClaudeJournalTranslator({ sink, fallbackIdPrefix: 'test' }) + const groupRows = () => + items.filter((item) => orcaClientMessageId(item.identity) === GROUP_ITEM_ID) + const agentsOf = (body: AgentJournalItemBody | undefined): NativeChatSubagentEntry[] => { + if (!body || body.kind !== 'message') { + return [] + } + const block = body.blocks.find( + (candidate): candidate is NativeChatSubagentGroupBlock => candidate.type === 'subagent-group' + ) + return block ? block.agents : [] + } + /** The last roster row written for one group, so a test can read a group that + * is no longer the live one. */ + const rosterIn = (groupId: string): NativeChatSubagentEntry[] => + agentsOf( + items.findLast((item) => orcaClientMessageId(item.identity) === `claude-subagents:${groupId}`) + ?.body + ) + const rosterOf = (turnUuid: string): NativeChatSubagentEntry[] => + rosterIn(`claude-session:${turnUuid}`) + const roster = (): NativeChatSubagentEntry[] => agentsOf(groupRows().at(-1)?.body) + const fallbackRows = (): AgentJournalItemBody[] => + items + .filter((item) => (orcaClientMessageId(item.identity) ?? '').startsWith('provider-frame:')) + .map((item) => item.body) + return { translator, groupRows, roster, rosterIn, rosterOf, fallbackRows } +} + +function userTurn(uuid: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { role: 'user', content: [{ type: 'text', text: 'go' }] } + } + } +} + +function systemFrame(subtype: string, fields: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { type: 'system', subtype, session_id: 'claude-session', ...fields } + } +} + +function spawnResult(uuid: string, toolUseId: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'user', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: toolUseId, content: 'done' }] + } + } + } +} + +function resultFrame() { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'result', + subtype: 'success', + session_id: 'claude-session', + uuid: 'result-1', + result: 'ok' + } + } +} + +describe('claude journal translation — subagents', () => { + it('rosters a spawned subagent and settles it on the spawn call result', () => { + const { translator, roster, fallbackRows } = harness() + translator.handle(userTurn('user-1')) + translator.handle( + systemFrame('task_started', { + task_id: 'task-1', + tool_use_id: 'toolu_1', + task_type: 'local_agent', + subagent_type: 'explorer', + description: 'Map the lane' + }) + ) + expect(roster()).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'Map the lane', state: 'working' }) + ]) + // The task frames stay status-chrome, so none of them prints an opcode row. + expect(fallbackRows()).toEqual([]) + translator.handle(spawnResult('user-2', 'toolu_1')) + expect(roster()).toEqual([expect.objectContaining({ state: 'completed' })]) + }) + + it('marks a child still working at turn end unverifiable', () => { + const { translator, roster } = harness() + translator.handle(userTurn('user-1')) + translator.handle( + systemFrame('task_started', { + task_id: 'task-1', + task_type: 'local_agent', + description: 'Map the lane' + }) + ) + translator.handle(resultFrame()) + expect(roster()).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) + + it('leaves a backgrounded child running past the end of its turn', () => { + const { translator, roster } = harness() + translator.handle(userTurn('user-1')) + translator.handle( + systemFrame('task_started', { + task_id: 'task-1', + tool_use_id: 'toolu_1', + task_type: 'local_agent', + description: 'Watch the build', + is_backgrounded: true + }) + ) + // A backgrounded spawn returns its tool result immediately; the child runs on. + translator.handle(spawnResult('user-2', 'toolu_1')) + translator.handle(resultFrame()) + expect(roster()).toEqual([expect.objectContaining({ state: 'working' })]) + translator.handle({ type: 'ended', sessionId: 'orca-session', reason: 'closed' }) + expect(roster()).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) + + it('keeps a backgrounded shell task out of the roster entirely', () => { + const { translator, groupRows } = harness() + translator.handle(userTurn('user-1')) + translator.handle( + systemFrame('task_started', { + task_id: 'task-bash', + tool_use_id: 'toolu_bash', + task_type: 'local_bash', + description: 'sleep 20', + is_backgrounded: true + }) + ) + translator.handle(resultFrame()) + expect(groupRows()).toEqual([]) + }) + + it('shows a subagent whose release announces no task frames, from its child traffic', () => { + const { translator, roster } = harness() + translator.handle(userTurn('user-1')) + translator.handle({ + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid: 'child-1', + session_id: 'claude-session', + parent_tool_use_id: 'toolu_1', + message: { role: 'assistant', content: [{ type: 'text', text: 'looking' }] } + } + }) + expect(roster()).toEqual([ + expect.objectContaining({ id: 'toolu_1', label: 'subagent', state: 'working' }) + ]) + }) + + it('settles the turn a new turn superseded, and leaves the new one running', () => { + const { translator, rosterOf } = harness() + translator.handle(userTurn('user-1')) + translator.handle( + systemFrame('task_started', { + task_id: 'task-1', + task_type: 'local_agent', + description: 'First turn' + }) + ) + // A second turn starts with no result frame for the first: the first turn + // ends here, and nothing else will ever name its group again. + translator.handle(userTurn('user-2')) + translator.handle( + systemFrame('task_started', { + task_id: 'task-2', + task_type: 'local_agent', + description: 'Second turn' + }) + ) + expect(rosterOf('user-1')).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + expect(rosterOf('user-2')).toEqual([expect.objectContaining({ state: 'working' })]) + }) + + it('does not let an unrelated turn end settle a child announced outside a turn', () => { + const { translator, rosterIn } = harness() + // No turn is live yet, so this child has no turn key to belong to. + translator.handle( + systemFrame('task_started', { + task_id: 'task-early', + task_type: 'local_agent', + description: 'Before the turn' + }) + ) + translator.handle(userTurn('user-1')) + translator.handle(resultFrame()) + expect(rosterIn('outside-turn')).toEqual([expect.objectContaining({ state: 'working' })]) + // The outcome still lands, which a latched `unverifiable` would have lost. + translator.handle( + systemFrame('task_updated', { task_id: 'task-early', patch: { status: 'completed' } }) + ) + expect(rosterIn('outside-turn')).toEqual([expect.objectContaining({ state: 'completed' })]) + }) + + it('settles a child left outside every turn when the session ends', () => { + const { translator, rosterIn } = harness() + translator.handle( + systemFrame('task_started', { + task_id: 'task-early', + task_type: 'local_agent', + description: 'Before the turn' + }) + ) + translator.handle(userTurn('user-1')) + translator.handle(resultFrame()) + translator.handle({ type: 'ended', sessionId: 'orca-session', reason: 'closed' }) + expect(rosterIn('outside-turn')).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) +}) diff --git a/src/main/claude/claude-structured-journal-translation.ts b/src/main/claude/claude-structured-journal-translation.ts index df8e8e67f53..759c2926763 100644 --- a/src/main/claude/claude-structured-journal-translation.ts +++ b/src/main/claude/claude-structured-journal-translation.ts @@ -11,9 +11,8 @@ import { claudeMessageBody, claudeMessageIdentity, claudeHasReplayContent, - claudeRecord, + claudeOutputEnvelope, claudeStreamingMessageBody, - claudeText, claudeThinkingIdentity, claudeThinkingText, claudeToolBody, @@ -29,16 +28,15 @@ import { claudeQuestionItems } from './claude-structured-prompt-items' import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' -import { readableProviderFrameText } from '../native-chat/agent-session-wire/unhandled-provider-frame' import { claudeProviderFrameActivity } from '../native-chat/agent-session-wire/provider-frame-activity' import { - CLAUDE_UNRENDERABLE_CONTENT_TEXT, + appendUnmodeledClaudeContent, claudeProviderFrameKind, claudeResultFailure, createClaudeProviderFrameFallback, - isModeledClaudeContent, isSettledClaudeResultKind } from './claude-structured-provider-fallback' +import { ClaudeSubagentRoster } from './claude-subagent-roster' import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' @@ -89,10 +87,16 @@ export function createClaudeJournalTranslator( const promptItems = new Map() const streamedBlocks = createClaudeStreamedBlockRegistry() let currentTurn: { sessionId: string; turnId: string } | null = null + const groupKeyOf = (turn: { sessionId: string; turnId: string } | null): string | null => + turn ? `${turn.sessionId}:${turn.turnId}` : null const providerFallback = createClaudeProviderFrameFallback( deps.sink, deps.fallbackIdPrefix ?? 'acquisition' ) + const subagents = new ClaudeSubagentRoster({ + sink: deps.sink, + currentGroupKey: () => groupKeyOf(currentTurn) + }) const streamedText = createClaudeStreamedTextCheckpoints({ ...(deps.coalesceMs === undefined ? {} : { coalesceMs: deps.coalesceMs }), ...(deps.schedule ? { schedule: deps.schedule } : {}), @@ -144,14 +148,10 @@ export function createClaudeJournalTranslator( return false } let changed = false - // User bubbles belong to the submitted message; SDK user frames carry echoes and tool results. - const outputEnvelope = - envelope.role === 'user' - ? { - ...envelope, - content: envelope.content.filter((part) => claudeRecord(part)?.type === 'tool_result') - } - : envelope + if (envelope.parentToolUseId) { + subagents.observeChildActivity(envelope.parentToolUseId) + } + const outputEnvelope = claudeOutputEnvelope(envelope) const body = claudeMessageBody(outputEnvelope) // The final frame of a streamed block lands on the block's identity, not its own uuid. const identity = @@ -180,6 +180,8 @@ export function createClaudeJournalTranslator( claudeToolIdentity(envelope.sessionId, result.toolUseId), claudeToolBody({ tool, result }) ) + // A spawn call's result is the parent turn's evidence its child finished. + subagents.observeToolResult(result.toolUseId, result.failed) // Tool inputs are only needed until their matching result arrives. tools.delete(result.toolUseId) changed = true @@ -192,21 +194,7 @@ export function createClaudeJournalTranslator( }) changed = true } - const unhandledContent = outputEnvelope.content.filter((part) => !isModeledClaudeContent(part)) - for (const part of unhandledContent) { - const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown' - providerFallback.append( - `message:${envelope.role}:content:${partType}`, - part, - readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT - ) - changed = true - } - // An empty user frame is a replay with nothing to show, not an unknown kind. - if (envelope.content.length === 0 && envelope.role === 'assistant') { - providerFallback.append(`message:${envelope.role}:empty`, message) - changed = true - } + changed = appendUnmodeledClaudeContent(providerFallback, outputEnvelope, message) || changed if ( envelope.role === 'user' && startsTurn && @@ -214,6 +202,9 @@ export function createClaudeJournalTranslator( message.parent_tool_use_id === null ) { if (currentTurn) { + // A new turn starting is the only end the previous one gets when its + // result never arrives; settling it later would sweep THIS turn. + subagents.settleTurn(groupKeyOf(currentTurn)) publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) } currentTurn = { sessionId: envelope.sessionId, turnId: envelope.uuid } @@ -254,6 +245,8 @@ export function createClaudeJournalTranslator( handle: (event) => { if (event.type === 'ended') { streamedText.flush() + // No event will ever settle a child once the provider is gone. + subagents.settleSession() if (currentTurn) { publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) currentTurn = null @@ -274,6 +267,9 @@ export function createClaudeJournalTranslator( promptItems.delete(event.promptKey) deps.sink.publish() } else if (event.type === 'message' && event.message.type === 'result') { + // The turn is over however it ended, so a foreground child still + // reported as working will never be settled by an event. + subagents.settleTurn(groupKeyOf(currentTurn)) if (currentTurn) { publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) currentTurn = null @@ -291,6 +287,9 @@ export function createClaudeJournalTranslator( providerFallback.append(kind, event.message, failure?.text) } } else if (event.type === 'message') { + // These frames stay `status-chrome`: the roster reads them here, and the + // fallback below still drops the raw frame instead of printing an opcode. + subagents.observeSystemFrame(event.message) const kind = claudeProviderFrameKind(event.message) if (!handleMessage(event.message, event.startsTurn === true)) { providerFallback.append(kind, event.message) @@ -310,6 +309,7 @@ export function createClaudeJournalTranslator( tools.clear() promptItems.clear() streamedBlocks.clear() + subagents.dispose() } } } diff --git a/src/main/claude/claude-structured-provider-fallback.ts b/src/main/claude/claude-structured-provider-fallback.ts index 2528ac027df..68aec07976b 100644 --- a/src/main/claude/claude-structured-provider-fallback.ts +++ b/src/main/claude/claude-structured-provider-fallback.ts @@ -4,8 +4,15 @@ import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from '../native-chat/agent-session-journal/journal-payload-bounds' import { CLAUDE_STREAM_JSON_FRAME_KINDS } from '../native-chat/agent-session-wire/claude-stream-json-frame-schema' -import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame' -import { claudeRecord, claudeText } from './claude-structured-item-translation' +import { + readableProviderFrameText, + unhandledProviderFrameJournalItem +} from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { + claudeRecord, + claudeText, + type ClaudeMessageEnvelope +} from './claude-structured-item-translation' export function claudeProviderFrameKind(message: Record): string { const type = claudeText(message.type) ?? 'unknown' @@ -123,3 +130,30 @@ export function createClaudeProviderFrameFallback( } } } + +export type ClaudeProviderFrameFallback = ReturnType + +/** Journal each content part this build does not model, plus the empty assistant + * frame a replay leaves behind (an empty USER frame is a replay with nothing to + * show, not an unknown kind). Returns whether anything was appended. */ +export function appendUnmodeledClaudeContent( + fallback: ClaudeProviderFrameFallback, + envelope: ClaudeMessageEnvelope, + message: Record +): boolean { + let changed = false + for (const part of envelope.content.filter((part) => !isModeledClaudeContent(part))) { + const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown' + fallback.append( + `message:${envelope.role}:content:${partType}`, + part, + readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT + ) + changed = true + } + if (envelope.content.length === 0 && envelope.role === 'assistant') { + fallback.append(`message:${envelope.role}:empty`, message) + changed = true + } + return changed +} diff --git a/src/main/claude/claude-subagent-group-row.test.ts b/src/main/claude/claude-subagent-group-row.test.ts new file mode 100644 index 00000000000..2d5ac56e22c --- /dev/null +++ b/src/main/claude/claude-subagent-group-row.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest' +import type { NativeChatSubagentEntry } from '../../shared/native-chat-types' +import { claudeSubagentGroupBody } from './claude-subagent-group-row' + +function entry(id: string, state: NativeChatSubagentEntry['state']): NativeChatSubagentEntry { + return { id, label: id, state, startedAt: 1 } +} + +/** The fallback sentence is the WHOLE row on mobile and paired web, which have + * no roster renderer, so these assertions are the entire contract there. */ +function sentence(agents: readonly NativeChatSubagentEntry[]): string { + const body = claudeSubagentGroupBody('turn-1', agents) + const block = body.kind === 'message' ? body.blocks[0] : undefined + return block && block.type === 'text' ? block.text : '' +} + +describe('claudeSubagentGroupBody fallback sentence', () => { + it('reads as a plain completion when every child completed', () => { + expect(sentence([entry('a', 'completed'), entry('b', 'completed')])).toBe('Ran 2 subagents') + }) + + it('keeps the singular noun for a lone child', () => { + expect(sentence([entry('a', 'completed')])).toBe('Ran 1 subagent') + expect(sentence([entry('a', 'working')])).toBe('Kicked off 1 subagent') + }) + + it('names an unverifiable child instead of claiming the group ran', () => { + expect(sentence([entry('a', 'completed'), entry('b', 'unverifiable')])).toBe( + 'Ran 2 subagents (1 unverifiable)' + ) + }) + + it('ranks the adverse outcome worst-first', () => { + expect( + sentence([entry('a', 'failed'), entry('b', 'unverifiable'), entry('c', 'completed')]) + ).toBe('Ran 3 subagents (1 failed)') + expect(sentence([entry('a', 'stopped'), entry('b', 'unverifiable')])).toBe( + 'Ran 2 subagents (1 stopped)' + ) + }) + + it('shows the adverse outcome while a sibling still works', () => { + expect( + sentence([entry('a', 'working'), entry('b', 'working'), entry('c', 'unverifiable')]) + ).toBe('Kicked off 3 subagents (1 unverifiable)') + }) + + it('leaves a benign settled state out of the sentence', () => { + expect(sentence([entry('a', 'idle'), entry('b', 'completed')])).toBe('Ran 2 subagents') + }) + + it('counts every child holding the worst adverse state', () => { + expect(sentence([entry('a', 'failed'), entry('b', 'failed'), entry('c', 'stopped')])).toBe( + 'Ran 3 subagents (2 failed)' + ) + }) +}) diff --git a/src/main/claude/claude-subagent-group-row.ts b/src/main/claude/claude-subagent-group-row.ts new file mode 100644 index 00000000000..58af6b6b346 --- /dev/null +++ b/src/main/claude/claude-subagent-group-row.ts @@ -0,0 +1,32 @@ +// The journal row one Claude spawn group writes: its durable identity and the +// body it revises in place. + +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import { subagentGroupFallbackText } from '../../shared/native-chat-subagent-summary' +import type { NativeChatSubagentEntry } from '../../shared/native-chat-types' + +/** Durable journal identity for the group's row — stable across revisions and + * across a restart, so replay finds the same row instead of appending a new one. */ +export function claudeSubagentGroupIdentity(groupId: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-subagents:${groupId}` } +} + +/** The roster row: the structured block plus the plain sentence an older client + * renders in its place. A message whose only block is the new variant would + * reach such a client with nothing it can draw. */ +export function claudeSubagentGroupBody( + groupId: string, + agents: readonly NativeChatSubagentEntry[] +): AgentJournalItemBody { + return { + kind: 'message', + role: 'system', + blocks: [ + { type: 'text', text: subagentGroupFallbackText(agents) }, + { type: 'subagent-group', groupId, agents: [...agents] } + ] + } +} diff --git a/src/main/claude/claude-subagent-id-aliases.test.ts b/src/main/claude/claude-subagent-id-aliases.test.ts new file mode 100644 index 00000000000..bffb549d173 --- /dev/null +++ b/src/main/claude/claude-subagent-id-aliases.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, it } from 'vitest' +import { ClaudeSubagentIds } from './claude-subagent-id-aliases' + +describe('ClaudeSubagentIds', () => { + it('resolves an aliased tool id to its task, and an unaliased id to itself', () => { + const ids = new ClaudeSubagentIds() + ids.alias('toolu_1', 'task-1') + expect(ids.canonical('toolu_1')).toBe('task-1') + expect(ids.canonical('toolu_unknown')).toBe('toolu_unknown') + }) + + it('remembers an exclusion under either of the ids that named it', () => { + const ids = new ClaudeSubagentIds() + ids.exclude('task-bash') + expect(ids.isExcluded('toolu_bash', 'task-bash')).toBe(true) + expect(ids.isExcluded(null, null)).toBe(false) + expect(ids.isExcluded('task-agent')).toBe(false) + }) + + it('drops the oldest alias past the bound and keeps the newest', () => { + const ids = new ClaudeSubagentIds() + for (let index = 0; index <= 512; index += 1) { + ids.alias(`toolu_${index}`, `task-${index}`) + } + // Evicted: the id now stands only for itself. + expect(ids.canonical('toolu_0')).toBe('toolu_0') + expect(ids.canonical('toolu_512')).toBe('task-512') + expect(ids.canonical('toolu_1')).toBe('task-1') + }) + + it('drops the oldest exclusion past the bound and keeps the newest', () => { + const ids = new ClaudeSubagentIds() + for (let index = 0; index <= 512; index += 1) { + ids.exclude(`task-${index}`) + } + expect(ids.isExcluded('task-0')).toBe(false) + expect(ids.isExcluded('task-512')).toBe(true) + expect(ids.isExcluded('task-1')).toBe(true) + }) + + it('does not retain oversized aliases or exclusions', () => { + const ids = new ClaudeSubagentIds() + const oversized = 'x'.repeat(513) + ids.alias(oversized, 'task-1') + ids.alias('tool-1', oversized) + ids.exclude(oversized) + expect(ids.canonical(oversized)).toBe(oversized) + expect(ids.canonical('tool-1')).toBe('tool-1') + expect(ids.isExcluded(oversized)).toBe(false) + }) + + it('forgets everything on clear', () => { + const ids = new ClaudeSubagentIds() + ids.alias('toolu_1', 'task-1') + ids.exclude('task-1') + ids.clear() + expect(ids.canonical('toolu_1')).toBe('toolu_1') + expect(ids.isExcluded('task-1')).toBe(false) + }) +}) diff --git a/src/main/claude/claude-subagent-id-aliases.ts b/src/main/claude/claude-subagent-id-aliases.ts new file mode 100644 index 00000000000..d06bc00bf8d --- /dev/null +++ b/src/main/claude/claude-subagent-id-aliases.ts @@ -0,0 +1,63 @@ +// Which Claude ids name the same subagent, and which name no subagent at all. +// +// Claude re-announces a resumed task under a NEW `tool_use_id` while `task_id` +// stays put, so tool ids are aliases of a canonical task id — a store keyed on +// the tool id would show the child twice after every resume. +// +// The exclusions matter just as much: `task_updated` carries no `task_type` and +// child traffic carries no task metadata at all, so the one announcement that +// said "this is a backgrounded shell, not an agent" has to be remembered or a +// later frame re-admits it. + +import { isBoundedClaudeTaskId } from './claude-background-task-tracker' + +/** Both maps are event-accumulated and nothing prunes them, so both are bounded. */ +const MAX_TOOL_USE_ALIASES = 512 +const MAX_EXCLUDED_IDS = 512 + +export class ClaudeSubagentIds { + private readonly canonicalByToolUse = new Map() + private readonly excluded = new Set() + + /** The task id a tool id stands for, or the id itself when nothing aliases it. */ + canonical(id: string): string { + return this.canonicalByToolUse.get(id) ?? id + } + + alias(toolUseId: string, taskId: string): void { + if (!isBoundedClaudeTaskId(toolUseId) || !isBoundedClaudeTaskId(taskId)) { + return + } + this.canonicalByToolUse.set(toolUseId, taskId) + while (this.canonicalByToolUse.size > MAX_TOOL_USE_ALIASES) { + const oldest = this.canonicalByToolUse.keys().next() + if (oldest.done || oldest.value === toolUseId) { + break + } + this.canonicalByToolUse.delete(oldest.value) + } + } + + exclude(id: string): void { + if (!isBoundedClaudeTaskId(id)) { + return + } + this.excluded.add(id) + while (this.excluded.size > MAX_EXCLUDED_IDS) { + const oldest = this.excluded.values().next() + if (oldest.done || oldest.value === id) { + break + } + this.excluded.delete(oldest.value) + } + } + + isExcluded(...ids: (string | null)[]): boolean { + return ids.some((id) => id !== null && this.excluded.has(id)) + } + + clear(): void { + this.canonicalByToolUse.clear() + this.excluded.clear() + } +} diff --git a/src/main/claude/claude-subagent-roster-state.ts b/src/main/claude/claude-subagent-roster-state.ts new file mode 100644 index 00000000000..2fa8971d856 --- /dev/null +++ b/src/main/claude/claude-subagent-roster-state.ts @@ -0,0 +1,75 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import type { NativeChatSubagentEntry } from '../../shared/native-chat-types' +import type { ClaudeSubagentTaskFrame } from './claude-subagent-task-frames' + +const MAX_INVOCATIONS_PER_SUBAGENT = 16 + +export type TrackedEntry = { + entry: NativeChatSubagentEntry + /** The only signal separating a child that dies with its turn from one told to + * outlive it. A turn-end sweep must leave a backgrounded child alone. */ + backgrounded: boolean + toolUseId: string | null + invocationIds: Set | null + /** Label before its ordinal suffix, so a later announcement can tell a + * provisional row from one that already carries the provider's own name. */ + labelBase: string +} + +export type RosterGroup = { + groupId: string + identity: AgentJournalItemIdentity + /** Insertion order is the display order; the map holds the state. */ + entries: Map + /** Lifetime admissions bound retained labels even when entries are removed. */ + admittedEntries: number + /** Labels remain reserved after removal or provisional-name replacement. */ + claimedLabels: Set + /** Last body written, so an idempotent replay writes no new revision. */ + lastSerialized: string | null +} + +// Invocation history stays with the entry, independent of the evicting alias cache. +export function applyClaudeSubagentInvocation( + tracked: TrackedEntry, + frame: ClaudeSubagentTaskFrame, + now: () => number +): boolean { + if (tracked.invocationIds === null) { + return false + } + const newInvocation = + frame.announcement && frame.toolUseId !== null && !tracked.invocationIds.has(frame.toolUseId) + if (newInvocation && frame.toolUseId) { + if (tracked.invocationIds.size >= MAX_INVOCATIONS_PER_SUBAGENT) { + tracked.invocationIds = null + tracked.entry = { ...tracked.entry, state: 'unverifiable', settledAt: now() } + return true + } + tracked.invocationIds.add(frame.toolUseId) + if (tracked.toolUseId !== null && tracked.toolUseId !== frame.toolUseId) { + tracked.backgrounded = frame.backgrounded ?? false + tracked.entry = { ...tracked.entry, state: frame.state ?? 'working', settledAt: undefined } + } + tracked.toolUseId = frame.toolUseId + } else if (tracked.toolUseId && frame.toolUseId && tracked.toolUseId !== frame.toolUseId) { + return false + } + if (tracked.toolUseId === null) { + tracked.toolUseId = frame.toolUseId + } + return true +} + +/** Two children can share a description; the ordinal keeps their rows apart + * without inventing a name the provider never sent. The probe is over the + * labels actually rendered, not a per-base counter: a generated `Audit 2` + * must not collide with a provider that names its own child `Audit 2`. */ +export function claimClaudeSubagentLabel(group: RosterGroup, base: string): string { + let candidate = base + for (let ordinal = 2; group.claimedLabels.has(candidate); ordinal++) { + candidate = `${base} ${ordinal}` + } + group.claimedLabels.add(candidate) + return candidate +} diff --git a/src/main/claude/claude-subagent-roster.test.ts b/src/main/claude/claude-subagent-roster.test.ts new file mode 100644 index 00000000000..da1f16d0a0e --- /dev/null +++ b/src/main/claude/claude-subagent-roster.test.ts @@ -0,0 +1,602 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { + NativeChatSubagentEntry, + NativeChatSubagentGroupBlock +} from '../../shared/native-chat-types' +import type { AgentSessionJournal } from '../native-chat/agent-session-journal/journal-store' +import { + createDeferredStructuredAgentSessionEventSink, + type StructuredAgentSessionEventSink +} from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { ClaudeSubagentRoster } from './claude-subagent-roster' + +const TURN_1 = 'claude-session:turn-1' + +function agentsOf(body: AgentJournalItemBody | undefined): NativeChatSubagentEntry[] { + if (!body || body.kind !== 'message') { + return [] + } + const block = body.blocks.find( + (candidate): candidate is NativeChatSubagentGroupBlock => candidate.type === 'subagent-group' + ) + return block ? block.agents : [] +} + +function isGroupRow(identity: AgentJournalItemIdentity, groupId: string): boolean { + return identity.provider === 'orca' && identity.clientMessageId === `claude-subagents:${groupId}` +} + +function harness(groupKey: string | null = TURN_1) { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const tombstones: AgentJournalItemIdentity[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: (identity) => tombstones.push(identity), + publish: vi.fn() + } + let clock = 1_000 + let key = groupKey + const roster = new ClaudeSubagentRoster({ + sink, + currentGroupKey: () => key, + now: () => (clock += 1) + }) + const roles = (): NativeChatSubagentEntry[] => agentsOf(items.at(-1)?.body) + /** The last row written for one group, so a test can read a row that is no + * longer the newest one. */ + const rolesIn = (groupId: string): NativeChatSubagentEntry[] => + agentsOf(items.findLast((item) => isGroupRow(item.identity, groupId))?.body) + return { + roster, + items, + tombstones, + roles, + rolesIn, + setGroupKey: (next: string | null) => { + key = next + } + } +} + +function system(subtype: string, fields: Record): Record { + return { type: 'system', subtype, session_id: 'claude-session', ...fields } +} + +function started(fields: Record): Record { + return system('task_started', { task_type: 'local_agent', ...fields }) +} + +describe('ClaudeSubagentRoster', () => { + it('builds the row from task_started, with the fallback sentence beside the block', () => { + const { roster, items, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Review the diff' }) + ) + expect(items).toHaveLength(1) + expect(items[0]?.identity).toEqual({ + provider: 'orca', + clientMessageId: 'claude-subagents:claude-session:turn-1' + }) + const body = items[0]?.body + expect(body?.kind === 'message' && body.blocks[0]).toEqual({ + type: 'text', + text: 'Kicked off 1 subagent' + }) + expect(roles()).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'Review the diff', state: 'working' }) + ]) + }) + + it('keeps a backgrounded shell task out of the roster', () => { + const { roster, items } = harness() + roster.observeSystemFrame( + system('task_started', { + task_id: 'task-bash', + tool_use_id: 'toolu_bash', + task_type: 'local_bash', + description: 'sleep 20', + is_backgrounded: true + }) + ) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-bash', patch: { status: 'running' } }) + ) + // Its own frames carry a tool_use_id, so only the excluded-id memory stops it. + roster.observeChildActivity('toolu_bash') + expect(items).toHaveLength(0) + }) + + it('never renders a task marked skip_transcript', () => { + const { roster, items } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-a', tool_use_id: 'toolu_a', skip_transcript: true }) + ) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-a', patch: { status: 'completed' } }) + ) + roster.observeChildActivity('toolu_a') + expect(items).toHaveLength(0) + }) + + it('drops a provisional row once an announcement says the task is not a subagent', () => { + const { roster, items, tombstones, roles } = harness() + roster.observeChildActivity('toolu_bash') + expect(roles()).toHaveLength(1) + roster.observeSystemFrame( + system('task_started', { + task_id: 'task-bash', + tool_use_id: 'toolu_bash', + task_type: 'local_bash' + }) + ) + expect(tombstones).toEqual([ + { provider: 'orca', clientMessageId: 'claude-subagents:claude-session:turn-1' } + ]) + expect(items).toHaveLength(1) + }) + + it('does not duplicate a resumed task re-announced under a new tool_use_id', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_first', description: 'Audit' }) + ) + roster.observeChildActivity('toolu_first') + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_second', description: 'Audit' }) + ) + roster.observeChildActivity('toolu_second') + expect(roles()).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'Audit', state: 'working' }) + ]) + }) + + it('adopts a row built from child traffic when the announcement finally names it', () => { + const { roster, roles } = harness() + roster.observeChildActivity('toolu_1') + expect(roles()).toEqual([expect.objectContaining({ id: 'toolu_1', label: 'subagent' })]) + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Explore' }) + ) + expect(roles()).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'Explore', state: 'working' }) + ]) + }) + + it('is idempotent: a repeated frame writes no new revision', () => { + const { roster, items } = harness() + const frame = started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Audit' }) + roster.observeSystemFrame(frame) + roster.observeSystemFrame(frame) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { status: 'running' } }) + ) + expect(items).toHaveLength(1) + }) + + it('latches a terminal state against a later live report', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' })) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { status: 'failed' } }) + ) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { status: 'running' } }) + ) + expect(roles()).toEqual([expect.objectContaining({ state: 'failed' })]) + }) + + it('ignores an update for a task it never rostered', () => { + const { roster, items } = harness() + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-unknown', patch: { status: 'running' } }) + ) + expect(items).toHaveLength(0) + }) + + it('disambiguates children that share a description', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Explore' })) + roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Explore' })) + expect(roles().map((agent) => agent.label)).toEqual(['Explore', 'Explore 2']) + }) + + describe('turn end', () => { + it('leaves a backgrounded child working and marks a foreground one unverifiable', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-fg', description: 'Foreground' })) + roster.observeSystemFrame( + started({ task_id: 'task-bg', description: 'Background', is_backgrounded: true }) + ) + roster.settleTurn(TURN_1) + expect(roles()).toEqual([ + expect.objectContaining({ label: 'Foreground', state: 'unverifiable' }), + expect.objectContaining({ label: 'Background', state: 'working' }) + ]) + }) + + it('never re-settles a child that already reported an outcome', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' })) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { status: 'completed' } }) + ) + roster.settleTurn(TURN_1) + expect(roles()).toEqual([expect.objectContaining({ state: 'completed' })]) + }) + + it('sweeps backgrounded children only when the provider itself is gone', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-bg', description: 'Background', is_backgrounded: true }) + ) + roster.settleTurn(TURN_1) + roster.settleSession() + expect(roles()).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) + }) + + describe('spawn tool result', () => { + it('settles a foreground child', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'toolu_1' })) + roster.observeToolResult('toolu_1', false) + expect(roles()).toEqual([expect.objectContaining({ state: 'completed' })]) + }) + + it('reports a failed spawn as failed', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'toolu_1' })) + roster.observeToolResult('toolu_1', true) + expect(roles()).toEqual([expect.objectContaining({ state: 'failed' })]) + }) + + it('ignores the immediate result a backgrounded spawn returns', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_1', is_backgrounded: true }) + ) + roster.observeToolResult('toolu_1', false) + expect(roles()).toEqual([expect.objectContaining({ state: 'working' })]) + }) + + it('ignores results for tools that are not spawn calls', () => { + const { roster, items } = harness() + roster.observeToolResult('toolu_read', false) + expect(items).toHaveLength(0) + }) + }) + + describe('label ordinals', () => { + it('never re-issues an ordinal a removed row gave up', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' })) + roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Audit' })) + // task-1 is re-announced as a shell task, so its row goes; reclaiming the + // ordinal it held would print a second 'Audit 2' beside the one still shown. + roster.observeSystemFrame( + system('task_started', { task_id: 'task-1', task_type: 'local_bash' }) + ) + roster.observeSystemFrame(started({ task_id: 'task-3', description: 'Audit' })) + expect(roles().map((agent) => agent.label)).toEqual(['Audit 2', 'Audit 3']) + }) + + it('never generates a label a provider-supplied one already took', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' })) + roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Audit' })) + // The provider's own name for the third child is the label the ordinal just + // generated for the second; a per-base counter would print it twice. + roster.observeSystemFrame(started({ task_id: 'task-3', description: 'Audit 2' })) + const labels = roles().map((agent) => agent.label) + expect(labels).toEqual(['Audit', 'Audit 2', 'Audit 2 2']) + expect(new Set(labels).size).toBe(labels.length) + }) + }) + + describe('child traffic for an id the CLI never declared', () => { + it('creates nothing once the CLI has announced any task at all', () => { + const { roster, items } = harness() + // A rejected announcement still proves this CLI declares what it spawns. + roster.observeSystemFrame( + system('task_started', { task_id: 'task-bash', task_type: 'local_bash' }) + ) + roster.observeChildActivity('toolu_never_announced') + expect(items).toHaveLength(0) + }) + + it('rejects an over-long provisional id instead of storing it as an entry id', () => { + const { roster, items } = harness() + // The announced path drops an id past `claudeTaskId`'s bound; the + // provisional one writes the same durable entry id, so it must too. + roster.observeChildActivity(`toolu_${'x'.repeat(512)}`) + expect(items).toHaveLength(0) + roster.observeChildActivity(`toolu_${'x'.repeat(500)}`) + expect(items).toHaveLength(1) + }) + + it('still rosters a subagent announced after a task the filter rejected', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + system('task_started', { task_id: 'task-bash', task_type: 'local_bash' }) + ) + // The gate closes the child-traffic fallback, never the announcement path. + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Explore' }) + ) + roster.observeChildActivity('toolu_1') + expect(roles()).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'Explore', state: 'working' }) + ]) + }) + + it('leaves a grandchild parented inside the sidechain out of the roster', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'toolu_1', description: 'Explore' }) + ) + roster.observeChildActivity('toolu_1') + // A tool the subagent itself ran: never announced, so never excluded either. + roster.observeChildActivity('toolu_inner') + expect(roles()).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'Explore', state: 'working' }) + ]) + }) + + it('still mints the provisional row for a release that announces no task', () => { + const { roster, roles } = harness() + roster.observeChildActivity('toolu_1') + // Not an announcement: the fallback path stays open for this release. + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-x', patch: { status: 'running' } }) + ) + roster.observeChildActivity('toolu_2') + expect(roles().map((agent) => agent.label)).toEqual(['subagent', 'subagent 2']) + }) + }) + + describe('groups that no later event can reach', () => { + it('loses contact with a group evicted past the bound', () => { + const { roster, rolesIn, setGroupKey } = harness('turn-0') + for (let index = 0; index < 33; index += 1) { + setGroupKey(`turn-${index}`) + roster.observeSystemFrame(started({ task_id: `task-${index}`, description: 'Audit' })) + } + expect(rolesIn('turn-0')).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + expect(rolesIn('turn-32')).toEqual([expect.objectContaining({ state: 'working' })]) + }) + + it('loses contact with a live child when the translator is disposed without an end', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-bg', description: 'Background', is_backgrounded: true }) + ) + roster.dispose() + expect(roles()).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) + + it('writes nothing on dispose when the session already settled', () => { + const { roster, items } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' })) + roster.settleSession() + const written = items.length + roster.dispose() + expect(items).toHaveLength(written) + }) + }) + + it('groups children outside any turn under their own row', () => { + const { roster, items } = harness(null) + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'Audit' })) + expect(items[0]?.identity).toEqual({ + provider: 'orca', + clientMessageId: 'claude-subagents:outside-turn' + }) + }) +}) + +describe('ClaudeSubagentRoster — the turn that is ending', () => { + it('leaves a child announced outside any turn alone when an unrelated turn ends', () => { + const { roster, rolesIn, setGroupKey } = harness(null) + roster.observeSystemFrame(started({ task_id: 'task-early', description: 'Early' })) + setGroupKey(TURN_1) + roster.observeSystemFrame(started({ task_id: 'task-turn', description: 'In turn' })) + roster.settleTurn(TURN_1) + expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'working' })]) + expect(rolesIn(TURN_1)).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + // `unverifiable` latches, so sweeping it above would have swallowed this. + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-early', patch: { status: 'completed' } }) + ) + expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'completed' })]) + }) + + it('sweeps the outside-turn group when a turn with no key of its own ends', () => { + const { roster, rolesIn } = harness(null) + roster.observeSystemFrame(started({ task_id: 'task-early', description: 'Early' })) + roster.settleTurn(null) + expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) + + it('still settles an outside-turn child once the session itself ends', () => { + const { roster, rolesIn, setGroupKey } = harness(null) + roster.observeSystemFrame(started({ task_id: 'task-early', description: 'Early' })) + setGroupKey(TURN_1) + roster.observeSystemFrame(started({ task_id: 'task-turn', description: 'In turn' })) + roster.settleTurn(TURN_1) + roster.settleSession() + expect(rolesIn('outside-turn')).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + }) + + it('sweeps the turn that ended, not whichever turn is live now', () => { + const { roster, rolesIn, setGroupKey } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'First turn' })) + setGroupKey('claude-session:turn-2') + roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Second turn' })) + // Turn 1's result lands after turn 2 has already begun. + roster.settleTurn(TURN_1) + expect(rolesIn(TURN_1)).toEqual([expect.objectContaining({ state: 'unverifiable' })]) + expect(rolesIn('claude-session:turn-2')).toEqual([ + expect.objectContaining({ state: 'working' }) + ]) + }) +}) + +describe('ClaudeSubagentRoster — through the real sink queue', () => { + it('lands every revision, not just the one that was already in flight', async () => { + const appended: AgentJournalItemBody[] = [] + let published = 0 + const journal = { + appendItem: async (_identity: AgentJournalItemIdentity, body: AgentJournalItemBody) => { + appended.push(body) + return { cursor: { epoch: 'e', sequence: appended.length } } + }, + appendTombstone: async () => ({ epoch: 'e', sequence: 0 }) + } as unknown as AgentSessionJournal + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ + journal, + fence: 1, + publish: () => { + published += 1 + } + }) + const roster = new ClaudeSubagentRoster({ sink: deferred.sink, currentGroupKey: () => TURN_1 }) + + // The first append is in flight while the rest are submitted, so a publish + // sharing the row's coalescing key would evict them. + roster.observeSystemFrame(started({ task_id: 'task-1', description: 'One' })) + roster.observeSystemFrame(started({ task_id: 'task-2', description: 'Two' })) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { status: 'completed' } }) + ) + const drained = await deferred.drained() + + expect(drained).toEqual({ ok: true }) + expect(agentsOf(appended.at(-1))).toEqual([ + expect.objectContaining({ id: 'task-1', label: 'One', state: 'completed' }), + expect.objectContaining({ id: 'task-2', label: 'Two', state: 'working' }) + ]) + expect(published).toBeGreaterThan(0) + }) +}) + +describe('ClaudeSubagentRoster — authoritative outcomes and retained budgets', () => { + it('accepts a notification after the foreground turn lost contact', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1' })) + roster.settleTurn(TURN_1) + roster.observeSystemFrame( + system('task_notification', { task_id: 'task-1', status: 'completed' }) + ) + expect(roles()).toEqual([expect.objectContaining({ state: 'completed' })]) + }) + + it('settles a background child from its notification without a task_updated', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', is_backgrounded: true })) + roster.settleTurn(TURN_1) + roster.observeSystemFrame(system('task_notification', { task_id: 'task-1', status: 'failed' })) + expect(roles()).toEqual([expect.objectContaining({ state: 'failed' })]) + }) + + it('bounds lifetime admissions when reclassification repeatedly removes entries', () => { + const { roster, items } = harness() + for (let i = 0; i < 100; i++) { + roster.observeSystemFrame(started({ task_id: `task-${i}`, description: `Agent ${i}` })) + roster.observeSystemFrame( + system('task_started', { task_id: `task-${i}`, task_type: 'local_bash' }) + ) + } + expect(items).toHaveLength(64) + }) +}) + +describe('ClaudeSubagentRoster — resumed invocation', () => { + it('reopens one canonical child on a new announcement without replaying old results', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' })) + roster.observeToolResult('first', false) + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'resumed', is_backgrounded: true }) + ) + expect(roles()).toEqual([expect.objectContaining({ id: 'task-1', state: 'working' })]) + expect(roles()[0].settledAt).toBeUndefined() + roster.observeSystemFrame( + system('task_notification', { task_id: 'task-1', tool_use_id: 'first', status: 'completed' }) + ) + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' })) + expect(roles()[0].state).toBe('working') + roster.observeSystemFrame( + system('task_notification', { + task_id: 'task-1', + tool_use_id: 'resumed', + status: 'completed' + }) + ) + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'resumed', is_backgrounded: true }) + ) + expect(roles()[0].state).toBe('completed') + }) +}) + +describe('ClaudeSubagentRoster — invocation fences', () => { + it('ignores a previous invocation tool result even without a background flag', () => { + const { roster, roles } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' })) + roster.observeToolResult('first', false) + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'next' })) + roster.observeToolResult('first', true) + expect(roles()[0].state).toBe('working') + roster.observeToolResult('next', false) + expect(roles()[0].state).toBe('completed') + }) + + it('does not treat an evicted alias as a new invocation', () => { + const { roster, rolesIn, setGroupKey } = harness() + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' })) + roster.observeToolResult('first', false) + setGroupKey('churn') + for (let i = 0; i < 513; i++) { + roster.observeSystemFrame( + system('task_updated', { task_id: `other-${i}`, tool_use_id: `tool-${i}` }) + ) + } + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'first' })) + expect(rolesIn(TURN_1)[0].state).toBe('completed') + }) + + it('bounds invocation history and refuses to reopen beyond the retained budget', () => { + const { roster, roles } = harness() + for (let i = 0; i < 20; i++) { + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: `tool-${i}` })) + if (i >= 16) { + expect(roles()[0].state).toBe('unverifiable') + } + roster.observeToolResult(`tool-${i}`, false) + } + roster.observeSystemFrame(started({ task_id: 'task-1', tool_use_id: 'tool-0' })) + expect(roles()[0].state).toBe('unverifiable') + }) +}) + +it('merges an explicit foreground patch without clearing on absent metadata', () => { + const { roster, roles } = harness() + roster.observeSystemFrame( + started({ task_id: 'task-1', tool_use_id: 'tool', is_backgrounded: true }) + ) + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { description: 'Audit' } }) + ) + roster.observeToolResult('tool', false) + expect(roles()[0].state).toBe('working') + roster.observeSystemFrame( + system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: false } }) + ) + roster.observeToolResult('tool', false) + expect(roles()[0].state).toBe('completed') +}) diff --git a/src/main/claude/claude-subagent-roster.ts b/src/main/claude/claude-subagent-roster.ts new file mode 100644 index 00000000000..34083f2ee8c --- /dev/null +++ b/src/main/claude/claude-subagent-roster.ts @@ -0,0 +1,388 @@ +// The Claude subagent roster: one journal row per turn that spawned children. +// +// Entries are built from `task_started`, never from child traffic: a +// BACKGROUNDED subagent emits no child frames at all, so a roster fed by +// `parent_tool_use_id` alone would leave every one of them an unlabelled row +// forever. Child traffic only creates an entry for CLI releases that announce +// no task frames. +// +// Claude re-announces a resumed task under a NEW `tool_use_id`, so `task_id` is +// the key and tool ids are aliases; keying on the tool id would duplicate the +// child on every resume. Outcomes latch within an invocation; a new spawn +// alias can reopen it, and authoritative evidence can correct lost contact. + +import { + canReplaceSubagentState, + isTerminalSubagentState +} from '../../shared/native-chat-subagent-summary' +import type { NativeChatSubagentEntry } from '../../shared/native-chat-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { isBoundedClaudeTaskId } from './claude-background-task-tracker' +import { claudeSubagentGroupBody, claudeSubagentGroupIdentity } from './claude-subagent-group-row' +import { ClaudeSubagentIds } from './claude-subagent-id-aliases' +import { readClaudeSubagentTaskFrame } from './claude-subagent-task-frames' +import { + applyClaudeSubagentInvocation, + claimClaudeSubagentLabel, + type RosterGroup, + type TrackedEntry +} from './claude-subagent-roster-state' + +/** Spawn-group rows kept live per session, and children per row. Both bound an + * event-accumulated map that no provider snapshot ever prunes. */ +const MAX_SUBAGENT_GROUPS = 32 +const MAX_SUBAGENTS_PER_GROUP = 64 + +/** The turn a group belongs to when Claude reports a task outside any turn. */ +const OUTSIDE_TURN = 'outside-turn' + +const UNLABELLED_AGENT = 'subagent' + +export type ClaudeSubagentRosterDeps = { + sink: StructuredAgentSessionEventSink + /** The turn that owns children spawned right now; null outside any turn. */ + currentGroupKey: () => string | null + now?: () => number +} + +export class ClaudeSubagentRoster { + private readonly groups = new Map() + /** Canonical id → the group holding its entry, so a late update for a child + * from an earlier turn revises that turn's row instead of the live one. */ + private readonly groupIdByEntry = new Map() + private readonly ids = new ClaudeSubagentIds() + /** Set by ANY `task_started`, including one the subagent filter rejects. Once + * this CLI has proven it declares its tasks, child traffic for an id it never + * announced is a nested tool or a grandchild, not a subagent. */ + private announcesTasks = false + private readonly now: () => number + + constructor(private readonly deps: ClaudeSubagentRosterDeps) { + this.now = deps.now ?? (() => Date.now()) + } + + /** Consume a `message:system:task_*` frame. Returns false when it is not one. */ + observeSystemFrame(message: Record): boolean { + const frame = readClaudeSubagentTaskFrame(message) + if (!frame) { + return false + } + this.announcesTasks ||= frame.announcement + if (frame.excluded) { + // Child traffic may already have built a provisional row under the tool id; + // the announcement is the first frame that says it is not a subagent. + for (const id of [frame.taskId, frame.toolUseId]) { + if (id !== null) { + this.ids.exclude(id) + this.remove(id) + } + } + return true + } + if (this.ids.isExcluded(frame.taskId, frame.toolUseId)) { + return true + } + if (frame.toolUseId) { + this.ids.alias(frame.toolUseId, frame.taskId) + } + const located = + this.locate(frame.taskId) ?? + (frame.toolUseId ? this.adopt(frame.toolUseId, frame.taskId) : null) + if (!located) { + if (frame.announcesSubagent) { + this.create( + frame.taskId, + frame.label, + frame.state ?? 'working', + frame.backgrounded ?? false, + frame.toolUseId + ) + } + return true + } + const tracked = located.group.entries.get(frame.taskId) + if (tracked && !applyClaudeSubagentInvocation(tracked, frame, this.now)) { + return true + } + this.revise(located.group, frame.taskId, { + label: frame.label, + state: frame.state, + backgrounded: frame.backgrounded + }) + return true + } + + /** + * A frame carrying `parent_tool_use_id` — the child's own traffic. It refreshes + * nothing on an announced child; it exists so a CLI release that sends no task + * frames still shows the subagent it is running. + */ + observeChildActivity(parentToolUseId: string): void { + const canonical = this.ids.canonical(parentToolUseId) + if (this.ids.isExcluded(parentToolUseId, canonical)) { + return + } + if (this.locate(canonical)) { + return + } + if (this.announcesTasks) { + // A nested Task, a workflow child, or a grandchild parented to a tool id + // inside the sidechain all reach here. This CLI announces what it spawns, + // so an id it never declared cannot be a subagent — and a row invented for + // one is unlabelled forever and can only ever end `unverifiable`. The + // bounded exclusion set cannot cover an id that was never announced. + return + } + if (!isBoundedClaudeTaskId(canonical)) { + // `claudeTaskId` rejects an over-long announced id rather than truncating + // it; a provisional id becomes the same durable entry key, so it cannot + // enter under a looser rule. + return + } + this.create(canonical, null, 'working', false, parentToolUseId) + } + + /** + * The parent turn's tool result for a spawn call. It settles a foreground + * child, whose result IS the turn's evidence the child finished. A backgrounded + * child's spawn call returns immediately while the child keeps running, so its + * result proves nothing and is ignored. + */ + observeToolResult(toolUseId: string, failed: boolean): void { + const canonical = this.ids.canonical(toolUseId) + const located = this.locate(canonical) + if ( + !located || + located.tracked.invocationIds === null || + located.tracked.backgrounded || + (located.tracked.toolUseId !== null && located.tracked.toolUseId !== toolUseId) + ) { + return + } + this.revise(located.group, canonical, { + label: null, + state: failed ? 'failed' : 'completed', + backgrounded: false + }) + } + + /** + * The parent turn ended. A foreground child still reported as working will + * never be settled by an event, so it becomes `unverifiable`: contact was + * lost, which is NOT evidence the child exited. A backgrounded child was + * explicitly told to outlive the turn and is left alone. + */ + settleTurn(groupKey: string | null): void { + // Only the group this key names. `OUTSIDE_TURN` belongs to no turn, so an + // unrelated turn ending is no evidence about a child announced outside it. + // `settleSession` reaches what no turn does. + this.sweep(this.groups.get(groupKey ?? OUTSIDE_TURN), false) + } + + /** The provider is gone. Nothing more will arrive for any child, backgrounded + * or not, so every one of them loses contact at once. */ + settleSession(): void { + for (const group of this.groups.values()) { + this.sweep(group, true) + } + } + + dispose(): void { + // Teardown paths reach here without an `ended` event, so a row still + // reporting `working` would have nothing left to revise it. A session that + // did settle first leaves every child terminal, so this writes nothing. + this.settleSession() + this.groups.clear() + this.groupIdByEntry.clear() + this.ids.clear() + this.announcesTasks = false + } + + private sweep(group: RosterGroup | undefined, includeBackgrounded: boolean): void { + if (!group) { + return + } + let changed = false + for (const [id, tracked] of group.entries) { + if (isTerminalSubagentState(tracked.entry.state)) { + continue + } + if (tracked.backgrounded && !includeBackgrounded) { + continue + } + group.entries.set(id, { + ...tracked, + entry: { ...tracked.entry, state: 'unverifiable', settledAt: this.now() } + }) + changed = true + } + if (changed) { + this.write(group) + } + } + + private create( + id: string, + label: string | null, + state: NativeChatSubagentEntry['state'], + backgrounded: boolean, + toolUseId: string | null + ): void { + const group = this.groupFor() + if (group.admittedEntries >= MAX_SUBAGENTS_PER_GROUP) { + return + } + group.admittedEntries += 1 + const now = this.now() + const labelBase = label ?? UNLABELLED_AGENT + group.entries.set(id, { + backgrounded, + toolUseId, + invocationIds: new Set(toolUseId ? [toolUseId] : []), + labelBase, + entry: { + id, + label: claimClaudeSubagentLabel(group, labelBase), + state, + startedAt: now, + ...(isTerminalSubagentState(state) ? { settledAt: now } : {}) + } + }) + this.groupIdByEntry.set(id, group.groupId) + this.write(group) + } + + private revise( + group: RosterGroup, + id: string, + change: { + label: string | null + state: NativeChatSubagentEntry['state'] | null + backgrounded: boolean | null + } + ): void { + const tracked = group.entries.get(id) + if (!tracked) { + return + } + const next: TrackedEntry = { + ...tracked, + backgrounded: change.backgrounded ?? tracked.backgrounded, + entry: { ...tracked.entry } + } + // A provisional row built from child traffic takes the real name the first + // announcement carries; an announced row keeps the name it was given. + if ( + change.label && + tracked.labelBase === UNLABELLED_AGENT && + change.label !== UNLABELLED_AGENT + ) { + next.labelBase = change.label + next.entry.label = claimClaudeSubagentLabel(group, change.label) + } + // Proven outcomes latch; lost contact can still receive a later verdict. + if (change.state && canReplaceSubagentState(tracked.entry.state, change.state)) { + next.entry.state = change.state + if (isTerminalSubagentState(change.state)) { + next.entry.settledAt = this.now() + } + } + group.entries.set(id, next) + this.write(group) + } + + /** Re-key a provisional entry from its tool id onto the canonical task id the + * announcement finally named, so the child does not appear twice. */ + private adopt(toolUseId: string, taskId: string): { group: RosterGroup } | null { + if (toolUseId === taskId) { + return null + } + const located = this.locate(toolUseId) + if (!located) { + return null + } + located.group.entries.delete(toolUseId) + located.group.entries.set(taskId, { + ...located.tracked, + entry: { ...located.tracked.entry, id: taskId } + }) + this.groupIdByEntry.delete(toolUseId) + this.groupIdByEntry.set(taskId, located.group.groupId) + return { group: located.group } + } + + private remove(id: string): void { + const located = this.locate(id) + if (!located) { + return + } + located.group.entries.delete(id) + this.groupIdByEntry.delete(id) + this.write(located.group) + } + + private locate(id: string): { group: RosterGroup; tracked: TrackedEntry } | null { + const groupId = this.groupIdByEntry.get(id) + const group = groupId === undefined ? undefined : this.groups.get(groupId) + const tracked = group?.entries.get(id) + return group && tracked ? { group, tracked } : null + } + + private groupFor(): RosterGroup { + const groupId = this.deps.currentGroupKey() ?? OUTSIDE_TURN + const existing = this.groups.get(groupId) + if (existing) { + return existing + } + const group: RosterGroup = { + groupId, + identity: claudeSubagentGroupIdentity(groupId), + entries: new Map(), + admittedEntries: 0, + claimedLabels: new Set(), + lastSerialized: null + } + this.groups.set(groupId, group) + while (this.groups.size > MAX_SUBAGENT_GROUPS) { + const oldest = this.groups.keys().next() + if (oldest.done || oldest.value === groupId) { + break + } + const evicted = this.groups.get(oldest.value) + // Once the group leaves the map nothing can reach its children again — + // not even a session sweep — so contact is lost here. + this.sweep(evicted, true) + for (const id of evicted?.entries.keys() ?? []) { + this.groupIdByEntry.delete(id) + } + this.groups.delete(oldest.value) + } + return group + } + + private write(group: RosterGroup): void { + const agents = [...group.entries.values()].map((tracked) => tracked.entry) + const options = { coalescingKey: `claude-subagents:${group.groupId}` } + if (agents.length === 0) { + // The row's last child turned out not to be a subagent. An empty roster is + // not a roster of nothing, so the row goes rather than reading "Ran 0". + if (group.lastSerialized !== null) { + group.lastSerialized = null + this.deps.sink.appendTombstone(group.identity, options) + this.deps.sink.publish() + } + return + } + const body = claudeSubagentGroupBody(group.groupId, agents) + const serialized = JSON.stringify(body) + if (serialized === group.lastSerialized) { + // Nothing changed — a duplicate delivery must not burn a revision. + return + } + group.lastSerialized = serialized + this.deps.sink.appendItem(group.identity, body, options) + // Publish keeps the sink's own coalescing slot: sharing the row's key makes + // each queued publish evict the append it was meant to flush. + this.deps.sink.publish() + } +} diff --git a/src/main/claude/claude-subagent-task-frames.test.ts b/src/main/claude/claude-subagent-task-frames.test.ts new file mode 100644 index 00000000000..230ef45e19c --- /dev/null +++ b/src/main/claude/claude-subagent-task-frames.test.ts @@ -0,0 +1,201 @@ +import { describe, expect, it } from 'vitest' +import { readClaudeSubagentTaskFrame } from './claude-subagent-task-frames' + +function system(subtype: string, fields: Record): Record { + return { type: 'system', subtype, session_id: 'claude-session', ...fields } +} + +describe('readClaudeSubagentTaskFrame', () => { + it('ignores frames that are not task frames', () => { + expect(readClaudeSubagentTaskFrame({ type: 'assistant', subtype: 'task_started' })).toBeNull() + expect(readClaudeSubagentTaskFrame(system('init', { task_id: 'task-1' }))).toBeNull() + expect(readClaudeSubagentTaskFrame(system('task_started', {}))).toBeNull() + expect(readClaudeSubagentTaskFrame(system('task_started', { task_id: '' }))).toBeNull() + }) + + describe('task_type triage', () => { + it('announces a local_agent task', () => { + const frame = readClaudeSubagentTaskFrame( + system('task_started', { + task_id: 'task-1', + tool_use_id: 'toolu_1', + task_type: 'local_agent', + subagent_type: 'code-reviewer', + description: 'Review the diff' + }) + ) + expect(frame).toMatchObject({ + taskId: 'task-1', + toolUseId: 'toolu_1', + label: 'Review the diff', + announcesSubagent: true, + excluded: false + }) + }) + + it('excludes a backgrounded shell command even though it carries a tool_use_id', () => { + const frame = readClaudeSubagentTaskFrame( + system('task_started', { + task_id: 'task-bash', + tool_use_id: 'toolu_bash', + task_type: 'local_bash', + description: 'sleep 20', + is_backgrounded: true + }) + ) + expect(frame).toMatchObject({ + taskId: 'task-bash', + toolUseId: 'toolu_bash', + announcesSubagent: false, + excluded: true + }) + }) + + it('excludes workflows and monitors', () => { + for (const taskType of ['local_workflow', 'monitor']) { + expect( + readClaudeSubagentTaskFrame( + system('task_started', { task_id: `task-${taskType}`, task_type: taskType }) + ) + ).toMatchObject({ announcesSubagent: false, excluded: true }) + } + }) + + it('caps a subagent_type label the way a description is capped', () => { + const frame = readClaudeSubagentTaskFrame( + system('task_started', { task_id: 'task-1', subagent_type: 'a'.repeat(900) }) + ) + // The roster stores this label verbatim, so nothing downstream bounds it. + expect(frame?.label).toHaveLength(512) + }) + + it('falls back to subagent_type only when the release sends no task_type', () => { + expect( + readClaudeSubagentTaskFrame( + system('task_started', { task_id: 'task-old', subagent_type: 'explorer' }) + ) + ).toMatchObject({ announcesSubagent: true, label: 'explorer' }) + expect( + readClaudeSubagentTaskFrame(system('task_started', { task_id: 'task-bare' })) + ).toMatchObject({ announcesSubagent: false, excluded: true }) + // A type this build does not recognise is not an agent on subagent_type's word. + expect( + readClaudeSubagentTaskFrame( + system('task_started', { + task_id: 'task-new', + task_type: 'local_something_new', + subagent_type: 'explorer' + }) + ) + ).toMatchObject({ announcesSubagent: false, excluded: true }) + }) + + it('excludes ambient housekeeping tasks', () => { + for (const suppression of [{ skip_transcript: true }, { ambient: true }]) { + expect( + readClaudeSubagentTaskFrame( + system('task_started', { + task_id: 'task-ambient', + task_type: 'local_agent', + subagent_type: 'watcher', + ...suppression + }) + ) + ).toMatchObject({ announcesSubagent: false, excluded: true }) + } + }) + }) + + describe('status', () => { + it('collapses every in-flight status to working', () => { + for (const status of ['pending', 'running', 'paused']) { + expect( + readClaudeSubagentTaskFrame( + system('task_updated', { task_id: 'task-1', patch: { status } }) + ) + ).toMatchObject({ state: 'working' }) + } + }) + + it('maps the settled statuses onto the carrier vocabulary', () => { + const mapped: [string, string][] = [ + ['completed', 'completed'], + ['failed', 'failed'], + ['killed', 'stopped'], + ['stopped', 'stopped'] + ] + for (const [status, state] of mapped) { + expect( + readClaudeSubagentTaskFrame( + system('task_updated', { task_id: 'task-1', patch: { status } }) + ) + ).toMatchObject({ state }) + } + }) + + it('reports no state for a status it cannot map', () => { + for (const status of ['__proto__', 'toString', 'invented', 7, null]) { + expect( + readClaudeSubagentTaskFrame( + system('task_updated', { task_id: 'task-1', patch: { status } }) + ) + ).toMatchObject({ state: null }) + } + }) + + it('treats progress as no lifecycle verdict', () => { + for (const subtype of ['task_progress']) { + expect( + readClaudeSubagentTaskFrame( + system(subtype, { task_id: 'task-1', status: 'completed', patch: { status: 'failed' } }) + ) + ).toMatchObject({ state: null }) + } + }) + }) + + it('reads the notification verdict from its top-level status', () => { + for (const state of ['completed', 'failed', 'stopped']) { + expect( + readClaudeSubagentTaskFrame( + system('task_notification', { + task_id: 'task-1', + status: state, + patch: { status: 'running' } + }) + ) + ).toMatchObject({ state }) + } + }) + + it('reads the backgrounded flag from the frame or its patch', () => { + expect( + readClaudeSubagentTaskFrame( + system('task_started', { + task_id: 'task-1', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + ).toMatchObject({ backgrounded: true }) + expect( + readClaudeSubagentTaskFrame( + system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: true } }) + ) + ).toMatchObject({ backgrounded: true }) + expect( + readClaudeSubagentTaskFrame(system('task_updated', { task_id: 'task-1', patch: {} })) + ).toMatchObject({ backgrounded: null }) + }) + + it('collapses a multi-line description into one bounded label', () => { + expect( + readClaudeSubagentTaskFrame( + system('task_updated', { + task_id: 'task-1', + patch: { description: ' audit\n the lockfile ' } + }) + ) + ).toMatchObject({ label: 'audit the lockfile' }) + }) +}) diff --git a/src/main/claude/claude-subagent-task-frames.ts b/src/main/claude/claude-subagent-task-frames.ts new file mode 100644 index 00000000000..e5f361fa8c4 --- /dev/null +++ b/src/main/claude/claude-subagent-task-frames.ts @@ -0,0 +1,123 @@ +// Claude's declarative task protocol, read as subagent roster events. +// +// `local_agent`, `local_workflow` and `local_bash` tasks all arrive on the same +// `message:system:task_*` channel and ALL carry a `tool_use_id`, so id presence +// discriminates nothing: filtering on it alone puts a backgrounded `sleep 20` in +// the subagent roster. `task_type` is the discriminator, with `subagent_type` +// covering CLI releases that predate it. + +import type { NativeChatSubagentState } from '../../shared/native-chat-types' +import { + classifyClaudeBackgroundTaskKind, + claudeTaskDescription, + claudeTaskId, + isBoundedClaudeTaskId +} from './claude-background-task-tracker' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +const TASK_SUBTYPES: ReadonlySet = new Set([ + 'task_started', + 'task_updated', + 'task_progress', + 'task_notification' +]) + +/** Provider status → the carrier's vocabulary. `killed` and `stopped` both mean + * the task was deliberately ended, which the carrier calls `stopped`; every + * in-flight status collapses to `working`. A Map, not an object, so a payload + * carrying `__proto__` as its status cannot resolve to an inherited value. */ +const TASK_STATES: ReadonlyMap = new Map([ + ['pending', 'working'], + ['running', 'working'], + ['paused', 'working'], + ['completed', 'completed'], + ['failed', 'failed'], + ['killed', 'stopped'], + ['stopped', 'stopped'] +] satisfies [string, NativeChatSubagentState][]) + +export type ClaudeSubagentTaskFrame = { + /** Canonical, resume-stable id — the roster key. */ + taskId: string + /** Re-minted when Claude re-announces a resumed task, so it is only an alias. */ + toolUseId: string | null + label: string | null + /** null when the frame reported no lifecycle status. */ + state: NativeChatSubagentState | null + backgrounded: boolean | null + /** Any `task_started`, subagent or not. Proof this CLI declares its tasks. */ + announcement: boolean + /** `task_started` for a task the roster should show. Only an announcement + * creates an entry: an update carries no `task_type`, so honouring one for an + * unknown id would roster whatever else shares this channel. */ + announcesSubagent: boolean + /** Ambient housekeeping, or a task that is not a subagent at all. Its ids must + * never reach the roster, by this frame or by later child traffic. */ + excluded: boolean +} + +/** True when the task Claude announced is a subagent rather than a backgrounded + * shell command or a workflow. */ +export function isClaudeSubagentTask(message: Record): boolean { + if (classifyClaudeBackgroundTaskKind(message.task_type) === 'agent') { + return true + } + // Releases predating `task_type` still name the child in `subagent_type`. A + // task_type Orca does not recognise is NOT covered: it is a type this build + // has no reason to believe is an agent. + return ( + (message.task_type === undefined || message.task_type === null) && + claudeText(message.subagent_type) !== null + ) +} + +function taskState(value: unknown): NativeChatSubagentState | null { + return typeof value === 'string' ? (TASK_STATES.get(value) ?? null) : null +} + +export function readClaudeSubagentTaskFrame( + message: Record +): ClaudeSubagentTaskFrame | null { + if (message.type !== 'system') { + return null + } + const subtype = claudeText(message.subtype) + if (!subtype || !TASK_SUBTYPES.has(subtype)) { + return null + } + const taskId = claudeTaskId(message) + if (!taskId) { + return null + } + const patch = claudeRecord(message.patch) + const toolUseId = claudeText(message.tool_use_id) ?? claudeText(patch?.tool_use_id) + const announcement = subtype === 'task_started' + // Housekeeping Claude runs for itself; the user never asked for it. + const suppressed = message.ambient === true || message.skip_transcript === true + const subagent = announcement && !suppressed && isClaudeSubagentTask(message) + return { + taskId, + toolUseId: toolUseId && isBoundedClaudeTaskId(toolUseId) ? toolUseId : null, + label: + claudeTaskDescription(message.description) ?? + claudeTaskDescription(patch?.description) ?? + // Bounded like a description: the roster stores whatever this returns. + (announcement ? (claudeTaskDescription(message.subagent_type) ?? null) : null), + // Notifications carry terminal evidence; progress carries usage only. + state: + subtype === 'task_notification' + ? taskState(message.status) + : announcement || subtype === 'task_updated' + ? taskState(patch?.status ?? message.status) + : null, + backgrounded: + typeof patch?.is_backgrounded === 'boolean' + ? patch.is_backgrounded + : typeof message.is_backgrounded === 'boolean' + ? message.is_backgrounded + : null, + announcement, + announcesSubagent: subagent, + excluded: announcement && !subagent + } +} diff --git a/src/main/codex-usage/codex-usage-rollup-projections.ts b/src/main/codex-usage/codex-usage-rollup-projections.ts index 0e03c2e80ee..783abb1d7fa 100644 --- a/src/main/codex-usage/codex-usage-rollup-projections.ts +++ b/src/main/codex-usage/codex-usage-rollup-projections.ts @@ -1,3 +1,4 @@ +import { highestUsageKey } from '../usage/highest-usage-key' import type { CodexUsageBreakdownKind, CodexUsageBreakdownRow, @@ -57,9 +58,8 @@ export function buildSummary( } } - const topModel = [...byModel.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null - const topProject = - [...byProject.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null + const topModel = highestUsageKey(byModel) + const topProject = highestUsageKey(byProject) return { scope, diff --git a/src/main/codex/codex-background-command-tracker.test.ts b/src/main/codex/codex-background-command-tracker.test.ts new file mode 100644 index 00000000000..abccb4e4a35 --- /dev/null +++ b/src/main/codex/codex-background-command-tracker.test.ts @@ -0,0 +1,128 @@ +import { describe, expect, it } from 'vitest' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { CodexBackgroundCommandTracker } from './codex-background-command-tracker' +import { createCodexJournalTranslator } from './codex-structured-journal-translation' +import type { CodexStructuredSessionEvent } from './codex-structured-session-state' + +function notification( + method: string, + params: Record +): Extract { + return { + type: 'notification', + sessionId: 'session', + threadId: 'root', + method, + params: { threadId: 'root', turnId: 'turn', ...params } + } +} + +function command(method: string, id = 'exec', threadId = 'root') { + return { + ...notification(method, { + item: { + type: 'commandExecution', + id, + command: 'sleep 30', + source: 'unifiedExecStartup', + status: method === 'item/completed' ? 'completed' : 'inProgress', + exitCode: method === 'item/completed' ? 0 : null + } + }), + threadId + } +} + +describe('persistent command ownership', () => { + it('preflights finite metadata capacity and admits work again after process completion', () => { + const tracker = new CodexBackgroundCommandTracker('root', 700) + const first = command('item/started', 'first') + const second = command('item/started', 'second') + expect(tracker.canObserve(first)).toBe(true) + tracker.observe(first) + expect(tracker.canObserve(second)).toBe(false) + expect(() => tracker.observe(second)).toThrow('not admitted') + expect(tracker.tasks()).toHaveLength(1) + expect(tracker.retainedMetadataBytes).toBeLessThanOrEqual(700) + tracker.observe(command('item/completed', 'first')) + expect(tracker.canObserve(second)).toBe(true) + tracker.observe(second) + expect(tracker.tasks()).toHaveLength(1) + expect(tracker.retainedMetadataBytes).toBeLessThanOrEqual(700) + tracker.clear() + expect(tracker.retainedMetadataBytes).toBe(0) + }) + + it('keeps the journal running across turn completion and accepts late output and exit', () => { + const rows: { key: string; body: AgentJournalItemBody }[] = [] + const translator = createCodexJournalTranslator({ + primaryThreadId: () => 'root', + sink: { + appendItem: (identity, body) => rows.push({ key: agentJournalItemKey(identity), body }), + appendTombstone: () => {}, + publish: () => {} + } + }) + const tracker = new CodexBackgroundCommandTracker('root') + const deliver = (event: Extract) => { + expect(translator.handle(event)).toEqual({ accepted: true }) + tracker.observe(event) + } + deliver(notification('turn/started', { turn: { id: 'turn' } })) + deliver(command('item/started')) + const originalKey = rows.find(({ body }) => body.kind === 'tool-call')?.key + deliver(notification('turn/completed', { turn: { id: 'turn' } })) + expect(rows.filter(({ body }) => body.kind === 'tool-call').map(({ body }) => body)).toEqual([ + expect.objectContaining({ state: 'running' }) + ]) + expect(tracker.tasks()).toHaveLength(1) + deliver( + notification('item/commandExecution/outputDelta', { itemId: 'exec', delta: 'late output' }) + ) + translator.flush() + expect(rows.at(-1)).toMatchObject({ key: originalKey, body: { state: 'running' } }) + deliver(command('item/completed')) + expect(rows.at(-1)).toMatchObject({ key: originalKey, body: { state: 'completed' } }) + expect(tracker.tasks()).toEqual([]) + translator.dispose() + }) + + it('counts child shells only after the child stops covering them, without resurrecting exits', () => { + const tracker = new CodexBackgroundCommandTracker('root') + tracker.observe(command('item/started', 'child-exec', 'child')) + tracker.observe( + notification('item/started', { + item: { + type: 'commandExecution', + id: 'poll', + source: 'unifiedExecInteraction', + status: 'inProgress' + } + }) + ) + expect(tracker.tasks(new Set(['child']))).toEqual([]) + expect(tracker.tasks()).toEqual([ + { id: 'codex-command:thread:child:child-exec', kind: 'command', description: 'sleep 30' } + ]) + tracker.observe(command('item/completed', 'child-exec', 'child')) + tracker.observe(command('item/started')) + tracker.observe(command('item/completed')) + tracker.observe(command('item/started')) + expect(tracker.tasks()).toEqual([]) + }) + + it('retains live commands while recycling bounded settled history', () => { + const tracker = new CodexBackgroundCommandTracker('root') + tracker.observe(command('item/started', 'long-lived')) + for (let index = 0; index < 300; index += 1) { + tracker.observe(command('item/started', `short-${index}`)) + tracker.observe(command('item/completed', `short-${index}`)) + } + expect(tracker.tasks()).toEqual([ + { id: 'codex-command:primary:long-lived', kind: 'command', description: 'sleep 30' } + ]) + tracker.clear() + expect(tracker.tasks()).toEqual([]) + }) +}) diff --git a/src/main/codex/codex-background-command-tracker.ts b/src/main/codex/codex-background-command-tracker.ts new file mode 100644 index 00000000000..becbb18a67c --- /dev/null +++ b/src/main/codex/codex-background-command-tracker.ts @@ -0,0 +1,150 @@ +import type { AgentSessionBackgroundTask } from '../../shared/agent-session-wire' +import type { CodexBackgroundTaskEvent } from './codex-background-task-frames' +import { codexCommandOutlivesTurn } from './codex-command-lifecycle' +import { readRecord, readString } from './codex-item-field-readers' +import { readCodexThreadItem } from './codex-structured-item-translation' +import { MAX_CODEX_ITEM_STREAM_METADATA_BYTES } from './codex-item-stream-retention' + +const MAX_SETTLED_COMMANDS = 128 +const MAX_DESCRIPTION_CHARS = 512 + +type Command = { threadId: string; task: AgentSessionBackgroundTask; bytes: number } + +/** Stays within the retained bound, so read-time qualification cannot outgrow admission. */ +function qualifiedDescription(label: string, description: string | undefined): string { + return (description ? `${label} — ${description}` : label).slice(0, MAX_DESCRIPTION_CHARS) +} + +export class CodexBackgroundCommandTracker { + private readonly commands = new Map() + private readonly settled = new Map() + private liveBytes = 0 + private settledBytes = 0 + + constructor( + private readonly primaryThreadId: string, + private readonly maxMetadataBytes = MAX_CODEX_ITEM_STREAM_METADATA_BYTES + ) {} + + get retainedMetadataBytes(): number { + return this.liveBytes + this.settledBytes + } + + canObserve(event: CodexBackgroundTaskEvent): boolean { + const parsed = this.parse(event) + return ( + !parsed || + parsed.completed || + this.commands.has(parsed.key) || + this.settled.has(parsed.key) || + this.liveBytes + parsed.command.bytes <= this.maxMetadataBytes + ) + } + + observe(event: CodexBackgroundTaskEvent): void { + const parsed = this.parse(event) + if (!parsed || this.settled.has(parsed.key)) { + return + } + const { key, command, completed } = parsed + const existing = this.commands.get(key) + if (completed) { + if (existing) { + this.liveBytes -= existing.bytes + this.commands.delete(key) + } + const bytes = Buffer.byteLength(key, 'utf8') + 256 + if (this.liveBytes + bytes <= this.maxMetadataBytes) { + this.settled.set(key, bytes) + this.settledBytes += bytes + } + this.trimSettled() + return + } + if (existing) { + return + } + if (this.liveBytes + command.bytes > this.maxMetadataBytes) { + throw new Error('Codex command metadata was not admitted before observation') + } + this.commands.set(key, command) + this.liveBytes += command.bytes + this.trimSettled() + } + + tasks( + coveredThreads?: ReadonlySet, + childLabel?: (threadId: string) => string | null + ): AgentSessionBackgroundTask[] { + return [...this.commands.values()] + .filter((command) => !coveredThreads?.has(command.threadId)) + .map(({ threadId, task }) => { + // The agent row carrying the child's name is gone by the time this row shows; + // unqualified it reads as a bare shell string with no owner. Resolved on read so + // a label registered after the command still lands. + const label = threadId === this.primaryThreadId ? null : childLabel?.(threadId) + return label + ? { ...task, description: qualifiedDescription(label, task.description) } + : task + }) + } + + clear(): void { + this.commands.clear() + this.settled.clear() + this.liveBytes = 0 + this.settledBytes = 0 + } + + private trimSettled(): void { + while ( + this.settled.size > MAX_SETTLED_COMMANDS || + this.retainedMetadataBytes > this.maxMetadataBytes + ) { + const oldest = this.settled.entries().next().value + if (!oldest) { + break + } + this.settled.delete(oldest[0]) + this.settledBytes -= oldest[1] + } + } + + private parse( + event: CodexBackgroundTaskEvent + ): { key: string; command: Command; completed: boolean } | null { + if (event.method !== 'item/started' && event.method !== 'item/completed') { + return null + } + const item = readCodexThreadItem(readRecord(event.params).item) + if (!item || !codexCommandOutlivesTurn(item)) { + return null + } + const key = JSON.stringify([event.threadId, item.id]) + const completed = event.method === 'item/completed' || item.status !== 'inProgress' + const description = readString(item, 'command') + ?.slice(0, MAX_DESCRIPTION_CHARS) + .replace(/\s+/g, ' ') + .trim() + const value = { + threadId: event.threadId, + task: { + id: + event.threadId === this.primaryThreadId + ? `codex-command:primary:${encodeURIComponent(item.id)}` + : `codex-command:thread:${encodeURIComponent(event.threadId)}:${encodeURIComponent(item.id)}`, + kind: 'command' as const, + ...(description ? { description } : {}) + } + } + return { + key, + completed, + command: { + ...value, + bytes: + Buffer.byteLength(key, 'utf8') + Buffer.byteLength(JSON.stringify(value), 'utf8') + 256 + } + } + } +} diff --git a/src/main/codex/codex-background-task-frames.ts b/src/main/codex/codex-background-task-frames.ts new file mode 100644 index 00000000000..a2e7a7e2151 --- /dev/null +++ b/src/main/codex/codex-background-task-frames.ts @@ -0,0 +1,72 @@ +import type { NativeChatSubagentState } from '../../shared/native-chat-types' +import { + codexSubagentLabel, + isCodexRootAgentActivity, + readCodexSubagentActivity +} from './codex-subagent-activity' +import { codexChildTurnState } from './codex-subagent-executions' +import { readRecord } from './codex-item-field-readers' +import { readCodexThreadItem } from './codex-structured-item-translation' +import { readCodexTurnId } from './codex-structured-thread-facts' + +export type CodexBackgroundTaskFrame = + | { + kind: 'subagent' + agentThreadId: string + label: string | null + parentTurnId: string | null | undefined + } + | { + kind: 'turn' + threadId: string + turnId: string + state: NativeChatSubagentState + } + +export type CodexBackgroundTaskEvent = { + method: string + threadId: string + params: unknown +} + +export function readCodexBackgroundTaskFrame( + event: CodexBackgroundTaskEvent, + primaryThreadId: string +): CodexBackgroundTaskFrame | null { + if (event.method === 'turn/started' || event.method === 'turn/completed') { + const turnId = readCodexTurnId(event.params) + if (turnId === null) { + return null + } + return { + kind: 'turn', + threadId: event.threadId, + turnId, + state: + event.method === 'turn/started' + ? 'working' + : codexChildTurnState(readRecord(readRecord(event.params).turn).status) + } + } + if (event.method !== 'item/started' && event.method !== 'item/completed') { + return null + } + const item = readCodexThreadItem(readRecord(event.params).item) + const activity = item && readCodexSubagentActivity(item) + if ( + !activity || + activity.agentThreadId === primaryThreadId || + isCodexRootAgentActivity(activity) + ) { + return null + } + return { + kind: 'subagent', + agentThreadId: activity.agentThreadId, + label: codexSubagentLabel(activity), + parentTurnId: + activity.kind === 'started' || activity.kind === 'interacted' + ? readCodexTurnId(event.params) + : undefined + } +} diff --git a/src/main/codex/codex-background-task-tracker.test.ts b/src/main/codex/codex-background-task-tracker.test.ts new file mode 100644 index 00000000000..47987fe3fc0 --- /dev/null +++ b/src/main/codex/codex-background-task-tracker.test.ts @@ -0,0 +1,280 @@ +import { describe, expect, it } from 'vitest' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' +import { + readCodexBackgroundTaskFrame, + type CodexBackgroundTaskEvent +} from './codex-background-task-frames' + +const PRIMARY = 'parent-thread' +const PARENT_TURN = 'parent-turn' +const CHILD = 'child-thread' +const CHILD_TURN = 'child-turn' + +function turn( + method: 'turn/started' | 'turn/completed', + threadId: string, + turnId: string, + status = 'completed' +): CodexBackgroundTaskEvent { + return { method, threadId, params: { threadId, turn: { id: turnId, status } } } +} + +function activity( + kind = 'started', + parentTurn = PARENT_TURN, + child = CHILD +): CodexBackgroundTaskEvent { + return { + method: 'item/started', + threadId: PRIMARY, + params: { + threadId: PRIMARY, + turnId: parentTurn, + item: { + type: 'subAgentActivity', + id: `activity-${kind}`, + kind, + agentThreadId: child, + agentPath: '/root/count_a' + } + } + } +} + +function runningChild(): CodexBackgroundTaskTracker { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(activity()) + return tracker +} + +function command(threadId = PRIMARY, method = 'item/started'): CodexBackgroundTaskEvent { + return { + method, + threadId, + params: { + threadId, + turnId: PARENT_TURN, + item: { + type: 'commandExecution', + id: 'exec-1', + processId: '71831', + source: 'unifiedExecStartup', + command: 'sleep 90', + status: method === 'item/started' ? 'inProgress' : 'completed' + } + } + } +} + +describe('readCodexBackgroundTaskFrame', () => { + it('reads activity as child metadata without inferring execution state', () => { + expect(readCodexBackgroundTaskFrame(activity('interacted'), PRIMARY)).toEqual({ + kind: 'subagent', + agentThreadId: CHILD, + label: 'count_a', + parentTurnId: PARENT_TURN + }) + }) + + it('reads a child turn with its own execution identity', () => { + expect(readCodexBackgroundTaskFrame(turn('turn/started', CHILD, CHILD_TURN), PRIMARY)).toEqual({ + kind: 'turn', + threadId: CHILD, + turnId: CHILD_TURN, + state: 'working' + }) + }) + + it('does not register the primary thread even when its activity path is missing', () => { + const event = activity('interacted', PARENT_TURN, PRIMARY) + ;(event.params as { item: { agentPath?: string } }).item.agentPath = undefined + expect(readCodexBackgroundTaskFrame(event, PRIMARY)).toBeNull() + }) +}) + +describe('CodexBackgroundTaskTracker child execution ownership', () => { + it('does not claim work from an activity item without a child turn', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(activity()) + tracker.observe(activity('interacted')) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.state).toBeNull() + }) + + it('reports an executing child only after the foreground turn ends', () => { + const tracker = runningChild() + expect(tracker.state).toBeNull() + expect(tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + supportsStopAll: false, + tasks: [{ id: `codex-agent:${CHILD}`, kind: 'agent', description: 'count_a' }] + }) + }) + + it('never settles a child when a primary turn ends', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + for (let index = 0; index < 300; index++) { + expect(tracker.observe(turn('turn/completed', PRIMARY, `later-${index}`))).toBe(false) + } + expect(tracker.state?.tasks).toHaveLength(1) + }) + + it.each(['completed', 'interrupted', 'failed'])( + 'settles on the matching child turn %s', + (status) => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.observe(turn('turn/completed', CHILD, CHILD_TURN, status))).toBe(true) + expect(tracker.state).toBeNull() + } + ) + + it('does not mistake late activity completion for the current child execution', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(activity('completed')) + expect(tracker.state?.tasks).toHaveLength(1) + }) + + it.each([PARENT_TURN, 'followup-parent'])( + 'reports follow-up work in %s using the new child turn', + (parentTurn) => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/started', PRIMARY, parentTurn)) + tracker.observe(activity('interacted', parentTurn)) + expect(tracker.state).toBeNull() + tracker.observe(turn('turn/started', CHILD, 'followup-child-turn')) + tracker.observe(turn('turn/completed', PRIMARY, parentTurn)) + expect(tracker.state?.tasks).toHaveLength(1) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(activity('completed')) + expect(tracker.state?.tasks).toHaveLength(1) + tracker.observe(turn('turn/completed', CHILD, 'followup-child-turn')) + expect(tracker.state).toBeNull() + } + ) + + it('keeps idle send_message activity out of the strip', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(activity('interacted', 'message-parent')) + tracker.observe(turn('turn/completed', PRIMARY, 'message-parent')) + expect(tracker.state).toBeNull() + }) + + it('does not invent another execution for a message to a working child', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(activity('interacted', 'message-parent')) + tracker.observe(turn('turn/completed', PRIMARY, 'message-parent')) + expect(tracker.state?.tasks).toHaveLength(1) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + expect(tracker.state).toBeNull() + }) + + it('retains completion delivered before child registration', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + tracker.observe(activity()) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.state).toBeNull() + }) + + it('publishes no extra state for duplicate owner or metadata events', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.observe(turn('turn/started', CHILD, CHILD_TURN))).toBe(false) + expect(tracker.observe({ ...activity(), method: 'item/completed' })).toBe(false) + expect(tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))).toBe(true) + expect(tracker.observe(turn('turn/completed', CHILD, CHILD_TURN))).toBe(false) + }) + + it('bounds retained child history while allowing repeated completed runs', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(activity()) + for (let index = 0; index < 300; index++) { + const id = `child-turn-${index}` + tracker.observe(turn('turn/started', CHILD, id)) + expect(tracker.state?.tasks).toHaveLength(1) + tracker.observe(turn('turn/completed', CHILD, id)) + expect(tracker.state).toBeNull() + } + }) + + it('clears the roster at session teardown', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.clear()).toBe(true) + expect(tracker.state).toBeNull() + expect(tracker.clear()).toBe(false) + }) +}) + +describe('CodexBackgroundTaskTracker command integration', () => { + it('keeps a primary shell visible after the turn until its own completion', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + tracker.observe(command()) + expect(tracker.state).toBeNull() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.state?.tasks).toEqual([ + { id: 'codex-command:primary:exec-1', kind: 'command', description: 'sleep 90' } + ]) + tracker.observe(command(PRIMARY, 'item/completed')) + expect(tracker.state).toBeNull() + }) + + it('reveals a child shell only after the child execution finishes', () => { + const tracker = runningChild() + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(command(CHILD)) + expect(tracker.state?.tasks).toHaveLength(1) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN, 'interrupted')) + expect(tracker.state?.tasks).toEqual([ + { + id: `codex-command:thread:${CHILD}:exec-1`, + kind: 'command', + description: 'count_a — sleep 90' + } + ]) + tracker.observe(command(CHILD, 'item/completed')) + expect(tracker.state).toBeNull() + }) + + it('leaves a primary shell unqualified', () => { + const tracker = runningChild() + tracker.observe(command(PRIMARY)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + expect(tracker.state?.tasks).toContainEqual({ + id: 'codex-command:primary:exec-1', + kind: 'command', + description: 'sleep 90' + }) + }) + + it('names a child shell whose label only arrives after the command', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(command(CHILD)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(activity()) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + expect(tracker.state?.tasks).toEqual([ + { + id: `codex-command:thread:${CHILD}:exec-1`, + kind: 'command', + description: 'count_a — sleep 90' + } + ]) + }) +}) diff --git a/src/main/codex/codex-background-task-tracker.ts b/src/main/codex/codex-background-task-tracker.ts new file mode 100644 index 00000000000..2918f087809 --- /dev/null +++ b/src/main/codex/codex-background-task-tracker.ts @@ -0,0 +1,96 @@ +import type { + AgentSessionBackgroundTask, + AgentSessionBackgroundTaskState +} from '../../shared/agent-session-wire' +import { + readCodexBackgroundTaskFrame, + type CodexBackgroundTaskEvent +} from './codex-background-task-frames' +import { CodexSubagentExecutions } from './codex-subagent-executions' +import { CodexBackgroundCommandTracker } from './codex-background-command-tracker' +import { boundSubagentField } from './codex-subagent-group-body' + +/** Projects the same child execution facts the durable roster consumes. */ +export class CodexBackgroundTaskTracker { + private primaryTurnId: string | null = null + private publishedFingerprint = '[]' + private publishedState: AgentSessionBackgroundTaskState | null = null + private readonly commands: CodexBackgroundCommandTracker + + constructor( + private readonly primaryThreadId: string, + private readonly executions = new CodexSubagentExecutions() + ) { + this.commands = new CodexBackgroundCommandTracker(primaryThreadId) + } + + get state(): AgentSessionBackgroundTaskState | null { + // Journal admission precedes observe; readers must not see its pending facts. + return this.publishedState + } + + canObserve(event: CodexBackgroundTaskEvent): boolean { + return this.commands.canObserve(event) + } + + observe(event: CodexBackgroundTaskEvent): boolean { + const itemEvent = event.method === 'item/started' || event.method === 'item/completed' + if (itemEvent) { + this.commands.observe(event) + } + const frame = readCodexBackgroundTaskFrame(event, this.primaryThreadId) + if (!frame) { + return itemEvent ? this.refresh() : false + } + if (frame.kind === 'subagent') { + this.executions.register(frame.agentThreadId, frame.label, frame.parentTurnId) + } else if (frame.threadId === this.primaryThreadId) { + if (frame.state === 'working') { + this.primaryTurnId = frame.turnId + } else if (frame.turnId === this.primaryTurnId) { + this.primaryTurnId = null + } + } else { + this.executions.observeTurn(frame.threadId, frame.turnId, frame.state) + } + return this.refresh() + } + + clear(): boolean { + this.executions.clear() + this.commands.clear() + this.primaryTurnId = null + return this.refresh() + } + + private tasks(): AgentSessionBackgroundTask[] { + if (this.primaryTurnId !== null) { + return [] + } + const children = this.executions.workingChildren() + const agents: AgentSessionBackgroundTask[] = children.map((child, index) => ({ + id: `codex-agent:${child.agentThreadId}`, + kind: 'agent', + ...(child.label ? { description: boundSubagentField(child.label, index) } : {}) + })) + return [ + ...agents, + ...this.commands.tasks(new Set(children.map((child) => child.agentThreadId)), (threadId) => + this.executions.label(threadId) + ) + ] + } + + private refresh(): boolean { + const tasks = this.tasks() + const fingerprint = JSON.stringify(tasks) + if (fingerprint === this.publishedFingerprint) { + return false + } + this.publishedFingerprint = fingerprint + this.publishedState = tasks.length + ? { state: 'monitoring', tasks, supportsStopAll: false } + : null + return true + } +} diff --git a/src/main/codex/codex-command-lifecycle.ts b/src/main/codex/codex-command-lifecycle.ts new file mode 100644 index 00000000000..dac8d7ddf97 --- /dev/null +++ b/src/main/codex/codex-command-lifecycle.ts @@ -0,0 +1,6 @@ +import type { CodexThreadItem } from './codex-structured-item-translation' + +/** Persistent exec has its own process-exit notification, independent of a turn. */ +export function codexCommandOutlivesTurn(item: CodexThreadItem): boolean { + return item.type === 'commandExecution' && item.source === 'unifiedExecStartup' +} diff --git a/src/main/codex/codex-item-stream-retention.ts b/src/main/codex/codex-item-stream-retention.ts new file mode 100644 index 00000000000..ace12f34667 --- /dev/null +++ b/src/main/codex/codex-item-stream-retention.ts @@ -0,0 +1,107 @@ +import { codexCommandOutlivesTurn } from './codex-command-lifecycle' +import { + MAX_CODEX_ITEM_STREAM_ITEM_BYTES, + MAX_CODEX_ITEM_STREAM_STATES +} from './codex-structured-item-stream-bounds' +import type { CodexItemStreamState } from './codex-structured-item-stream-contracts' + +// Preserve the previous metadata ceiling while letting small live commands share it. +export const MAX_CODEX_ITEM_STREAM_METADATA_BYTES = + MAX_CODEX_ITEM_STREAM_STATES * MAX_CODEX_ITEM_STREAM_ITEM_BYTES + +type RetainedState = { state: CodexItemStreamState; bytes: number; persistent: boolean } + +export class CodexItemStreamRetention { + private readonly states = new Map() + private bytes = 0 + private persistentBytes = 0 + private persistentCount = 0 + + constructor(private readonly maxBytes = MAX_CODEX_ITEM_STREAM_METADATA_BYTES) {} + + get retainedBytes(): number { + return this.bytes + } + + get size(): number { + return this.states.size + } + + get persistentSize(): number { + return this.persistentCount + } + + get overCapacity(): boolean { + return ( + this.bytes > this.maxBytes || + this.states.size - this.persistentCount > MAX_CODEX_ITEM_STREAM_STATES + ) + } + + get(key: string): CodexItemStreamState | undefined { + return this.states.get(key)?.state + } + + isPersistent(key: string): boolean { + return this.states.get(key)?.persistent === true + } + + canRetain(key: string, state: CodexItemStreamState): boolean { + const previous = this.states.get(key) + return ( + this.persistentBytes - + (previous?.persistent ? previous.bytes : 0) + + this.stateBytes(key, state) <= + this.maxBytes + ) + } + + retain(key: string, state: CodexItemStreamState): boolean { + if (!this.canRetain(key, state)) { + return false + } + this.forget(key) + const bytes = this.stateBytes(key, state) + const persistent = codexCommandOutlivesTurn(state.item) + this.states.set(key, { state, bytes, persistent }) + this.bytes += bytes + if (persistent) { + this.persistentBytes += bytes + this.persistentCount += 1 + } + return true + } + + oldestEvictable(): string | undefined { + for (const [key, entry] of this.states) { + if (!entry.persistent) { + return key + } + } + return undefined + } + + forget(key: string): void { + const entry = this.states.get(key) + if (!entry) { + return + } + this.bytes -= entry.bytes + if (entry.persistent) { + this.persistentBytes -= entry.bytes + this.persistentCount -= 1 + } + this.states.delete(key) + } + + clear(): void { + this.states.clear() + this.bytes = 0 + this.persistentBytes = 0 + this.persistentCount = 0 + } + + private stateBytes(key: string, state: CodexItemStreamState): number { + return Buffer.byteLength(key, 'utf8') + Buffer.byteLength(JSON.stringify(state), 'utf8') + 256 + } +} diff --git a/src/main/codex/codex-persistent-command-retention.test.ts b/src/main/codex/codex-persistent-command-retention.test.ts new file mode 100644 index 00000000000..2acb357cd71 --- /dev/null +++ b/src/main/codex/codex-persistent-command-retention.test.ts @@ -0,0 +1,219 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { CodexBackgroundCommandTracker } from './codex-background-command-tracker' +import { + CodexItemStreamRetention, + MAX_CODEX_ITEM_STREAM_METADATA_BYTES +} from './codex-item-stream-retention' +import { CodexJournalItems } from './codex-structured-journal-items' +import { settleCodexJournalTurn } from './codex-structured-journal-settlement' + +function command(threadId: string, id: string, method = 'item/started') { + return { + threadId, + method, + params: { + turnId: 'turn', + item: { + type: 'commandExecution', + id, + source: 'unifiedExecStartup', + command: `sleep 30 # ${threadId}/${id}`, + cwd: '/workspace', + status: method === 'item/completed' ? 'completed' : 'inProgress', + ...(method === 'item/completed' ? { exitCode: 0, aggregatedOutput: 'BEFORE\nAFTER\n' } : {}) + } + } + } +} + +function fixture(maxMetadataBytes?: number) { + const rows = new Map() + const scheduled = new Set<() => void>() + const sink = { + appendItem: ( + identity: Parameters[0], + body: AgentJournalItemBody + ) => { + rows.set(agentJournalItemKey(identity), body) + }, + appendTombstone() {}, + publish() {} + } + const items = new CodexJournalItems( + { + sink, + maxMetadataBytes, + schedule: (run) => { + scheduled.add(run) + return () => { + scheduled.delete(run) + } + } + }, + () => 'turn', + () => {} + ) + return { items, sink, rows, scheduled } +} + +describe('persistent command retention', () => { + it('does not rebuild unchanged persistent output on every later lifecycle flush', () => { + const { items } = fixture() + const event = command('root', 'quiet') + items.handle(event) + items.streams.handle('root', 'item/commandExecution/outputDelta', { + itemId: 'quiet', + delta: 'retained-prefix' + }) + items.streams.flush() + const originalJoin = Array.prototype.join + let retainedJoins = 0 + const spy = vi + .spyOn(Array.prototype, 'join') + .mockImplementation(function (this: unknown[], separator) { + if (this[0] === 'retained-prefix') { + retainedJoins += 1 + } + return originalJoin.call(this, separator) + }) + try { + for (let index = 0; index < 100; index += 1) { + items.streams.flush() + } + } finally { + spy.mockRestore() + items.dispose() + } + expect(retainedJoins).toBe(0) + }) + + it('retains 448 live commands through completed turns, late output, and process completion', () => { + const { items, sink, rows, scheduled } = fixture() + const tracker = new CodexBackgroundCommandTracker('thread-0') + const events = Array.from({ length: 7 }, (_, thread) => + Array.from({ length: 64 }, (_, index) => command(`thread-${thread}`, `exec-${index}`)) + ).flat() + for (const event of events) { + expect(tracker.canObserve(event)).toBe(true) + expect(items.handle(event)).toMatchObject({ admission: { accepted: true } }) + tracker.observe(event) + expect( + items.streams.handle(event.threadId, 'item/commandExecution/outputDelta', { + turnId: 'turn', + itemId: event.params.item.id, + delta: 'BEFORE\n' + }).admission + ).toEqual({ accepted: true }) + } + for (let thread = 0; thread < 7; thread += 1) { + expect( + settleCodexJournalTurn({ + sessionId: 'session', + threadId: `thread-${thread}`, + turnId: 'turn', + sink, + streams: items.streams, + activeItems: items.activeItems + }) + ).toEqual({ accepted: true }) + } + expect(items.activeItems.size).toBe(448) + expect(items.streams.persistentCount).toBe(448) + expect(tracker.tasks()).toHaveLength(448) + expect(tracker.retainedMetadataBytes).toBeLessThan(256 * 1024) + for (const event of events) { + items.streams.handle(event.threadId, 'item/commandExecution/outputDelta', { + turnId: 'turn', + itemId: event.params.item.id, + delta: 'AFTER\n' + }) + } + expect(items.streams.flush()).toBe(true) + for (const event of events) { + const key = agentJournalItemKey({ + provider: 'orca', + clientMessageId: `codex-item:${event.threadId}:${event.params.item.id}` + }) + expect(rows.get(key)).toMatchObject({ + state: 'running', + input: { command: event.params.item.command, cwd: '/workspace' }, + output: { head: 'BEFORE\nAFTER\n' } + }) + const completed = command(event.threadId, event.params.item.id, 'item/completed') + expect(items.handle(completed)).toMatchObject({ admission: { accepted: true } }) + tracker.observe(completed) + expect(rows.get(key)).toMatchObject({ + state: 'completed', + output: { head: 'BEFORE\nAFTER\n' } + }) + expect(items.streams.snapshot(event.threadId, event.params.item.id)).toBeNull() + } + expect(items.activeItems.size).toBe(0) + expect(items.streams.persistentCount).toBe(0) + expect(tracker.tasks()).toEqual([]) + expect(tracker.retainedMetadataBytes).toBeLessThan(64 * 1024) + items.dispose() + tracker.clear() + expect(tracker.retainedMetadataBytes).toBe(0) + expect(scheduled.size).toBe(0) + }) + + it('rejects command metadata exhaustion before appending or evicting live state and frees it on completion', () => { + const { items, rows } = fixture(800) + const first = command('root', 'first') + const second = command('root', 'second') + expect(items.handle(first)).toMatchObject({ admission: { accepted: true } }) + const prior = [...rows] + expect(items.handle(second)).toMatchObject({ admission: { accepted: false, reason: 'failed' } }) + expect([...rows]).toEqual(prior) + expect(items.activeItems.size).toBe(1) + expect(items.handle(command('root', 'first', 'item/completed'))).toMatchObject({ + admission: { accepted: true } + }) + expect(items.handle(second)).toMatchObject({ admission: { accepted: true } }) + items.dispose() + }) + + it('accounts metadata bytes instead of interpreting the item count as liveness', () => { + const retention = new CodexItemStreamRetention() + for (let index = 0; index < 448; index += 1) { + const item = command('root', `exec-${index}`).params.item + expect( + retention.retain(item.id, { + item, + identity: { provider: 'orca', clientMessageId: item.id } + }) + ).toBe(true) + } + expect(retention.size).toBe(448) + expect(retention.retainedBytes).toBeLessThan(256 * 1024) + expect(retention.retainedBytes).toBeLessThan(MAX_CODEX_ITEM_STREAM_METADATA_BYTES) + expect(retention.overCapacity).toBe(false) + expect(retention.oldestEvictable()).toBeUndefined() + retention.clear() + expect(retention.retainedBytes).toBe(0) + expect(retention.persistentSize).toBe(0) + }) + + it('retains startup provenance when large command metadata is bounded', () => { + const { items, sink } = fixture() + const event = command('root', 'large') + event.params.item.command = 'x'.repeat(128 * 1024) + expect(items.handle(event)).toMatchObject({ admission: { accepted: true } }) + expect( + settleCodexJournalTurn({ + sessionId: 'session', + threadId: 'root', + turnId: 'turn', + sink, + streams: items.streams, + activeItems: items.activeItems + }) + ).toEqual({ accepted: true }) + expect(items.activeItems.size).toBe(1) + expect(items.streams.persistentCount).toBe(1) + items.dispose() + }) +}) diff --git a/src/main/codex/codex-session-backfill-scan-dates.test.ts b/src/main/codex/codex-session-backfill-scan-dates.test.ts index f9bf4530733..8e7001c636c 100644 --- a/src/main/codex/codex-session-backfill-scan-dates.test.ts +++ b/src/main/codex/codex-session-backfill-scan-dates.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { compareCodexSessionBackfillDates, expandCodexSessionBackfillDatesThroughToday, @@ -9,6 +9,7 @@ import { parseCodexSessionBackfillDates, subtractCodexSessionBackfillDates } from './codex-session-backfill-scan-dates' +import type { CodexSessionBackfillDate } from './codex-session-backfill-types' describe('codex session backfill scan dates', () => { it('reads UTC parts so a local evening never lands on the wrong directory', () => { @@ -100,3 +101,63 @@ describe('codex session backfill scan dates', () => { ).toBeNull() }) }) + +describe('bounded backfill range construction', () => { + it('does not allocate rejected dates for a decades-old pending marker', () => { + const advance = vi.spyOn(Date.prototype, 'setUTCDate') + try { + expect( + expandCodexSessionBackfillDatesThroughToday( + [['2000', '01', '01']], + ['2026', '09', '07'], + 31 + ) + ).toBeNull() + expect(advance).not.toHaveBeenCalled() + } finally { + advance.mockRestore() + } + }) + + it('keeps exact, fractional, leap-day and future-clock bounds', () => { + const dates = [['2024', '02', '28']] as [string, string, string][] + expect(expandCodexSessionBackfillDatesThroughToday(dates, ['2024', '03', '01'], 3)).toEqual([ + ['2024', '02', '28'], + ['2024', '02', '29'], + ['2024', '03', '01'] + ]) + expect(expandCodexSessionBackfillDatesThroughToday(dates, ['2024', '03', '01'], 2.5)).toBeNull() + expect( + expandCodexSessionBackfillDatesThroughToday([['2024', '03', '01']], ['2024', '02', '28'], 3) + ).toEqual(expandCodexSessionBackfillDatesThroughToday(dates, ['2024', '03', '01'], 3)) + }) + + // The arithmetic cardinality gate must admit and reject exactly what enumerating the range + // would, on every calendar edge that has ever broken a day count: leap days, century rules, + // year rollover, and the DST switches the UTC-only arithmetic has to stay indifferent to. + it.each<[string, CodexSessionBackfillDate, CodexSessionBackfillDate]>([ + ['leap February', ['2024', '02', '27'], ['2024', '03', '02']], + ['non-leap February', ['2023', '02', '27'], ['2023', '03', '02']], + ['US spring-forward', ['2024', '03', '09'], ['2024', '03', '11']], + ['US fall-back', ['2024', '11', '02'], ['2024', '11', '04']], + ['EU spring-forward', ['2025', '03', '29'], ['2025', '03', '31']], + ['southern-hemisphere DST', ['2025', '04', '05'], ['2025', '04', '07']], + ['year rollover', ['2024', '12', '30'], ['2025', '01', '02']], + ['leap century', ['1999', '12', '31'], ['2000', '01', '02']], + ['non-leap century', ['2100', '02', '27'], ['2100', '03', '02']], + ['30-day month end', ['2026', '04', '29'], ['2026', '05', '02']], + ['single day', ['2026', '09', '07'], ['2026', '09', '07']] + ])('matches the enumerated range at the %s cap boundary', (_label, from, to) => { + const start = new Date(Date.UTC(Number(from[0]), Number(from[1]) - 1, Number(from[2]))) + const end = new Date(Date.UTC(Number(to[0]), Number(to[1]) - 1, Number(to[2]))) + const enumerated = getCodexSessionBackfillDatesBetween(start, end) + const pending = [from] + + expect(expandCodexSessionBackfillDatesThroughToday(pending, to, enumerated.length)).toEqual( + enumerated + ) + expect( + expandCodexSessionBackfillDatesThroughToday(pending, to, enumerated.length - 1) + ).toBeNull() + }) +}) diff --git a/src/main/codex/codex-session-backfill-scan-dates.ts b/src/main/codex/codex-session-backfill-scan-dates.ts index ca4e1d311f7..ac9a07d3e79 100644 --- a/src/main/codex/codex-session-backfill-scan-dates.ts +++ b/src/main/codex/codex-session-backfill-scan-dates.ts @@ -99,8 +99,13 @@ export function expandCodexSessionBackfillDatesThroughToday( return [] } const bounds = mergeCodexSessionBackfillDates(dates, [today]) - const range = getCodexSessionBackfillDatesBetween(toUtcDate(bounds[0]), toUtcDate(bounds.at(-1)!)) - return range.length > maxDates ? null : range + const first = toUtcDate(bounds[0]) + const last = toUtcDate(bounds.at(-1)!) + const dateCount = (last.getTime() - first.getTime()) / 86_400_000 + 1 + if (dateCount > maxDates) { + return null + } + return getCodexSessionBackfillDatesBetween(first, last) } function toUtcDate([year, month, day]: readonly string[]): Date { diff --git a/src/main/codex/codex-structured-item-stream-bounds.ts b/src/main/codex/codex-structured-item-stream-bounds.ts index 117b9b3a911..46104c54360 100644 --- a/src/main/codex/codex-structured-item-stream-bounds.ts +++ b/src/main/codex/codex-structured-item-stream-bounds.ts @@ -30,6 +30,7 @@ export function boundStreamItem(item: Record): Record void + readonly persistentCount: number + canTrack: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => boolean + track: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => boolean handle: ( threadId: string, method: string, diff --git a/src/main/codex/codex-structured-item-streams.ts b/src/main/codex/codex-structured-item-streams.ts index a765f8339da..e67263f133d 100644 --- a/src/main/codex/codex-structured-item-streams.ts +++ b/src/main/codex/codex-structured-item-streams.ts @@ -1,5 +1,6 @@ import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' import { createAgentSessionDeltaCoalescer } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' +import { CodexItemStreamRetention } from './codex-item-stream-retention' import { codexJournalItem, codexStreamingJournalItem, @@ -10,7 +11,6 @@ import { MAX_CODEX_ITEM_STREAM_PENDING_PATCHES, MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES, MAX_CODEX_ITEM_STREAM_RETAINED_BYTES, - MAX_CODEX_ITEM_STREAM_STATES, boundStreamItem, pendingPatchBytes } from './codex-structured-item-stream-bounds' @@ -47,8 +47,9 @@ export { export function createCodexStructuredItemStreams( deps: CodexItemStreamDeps ): CodexStructuredItemStreams { - const states = new Map() + const states = new CodexItemStreamRetention(deps.maxMetadataBytes) const checkpointLengths = new Map() + const pendingCheckpoints = new Set() // Patch updates are authoritative item snapshots. Keep the latest rejected // snapshot until the journal admits it; unlike streamed deltas, there is no // coalescer timer to retry these events for us. @@ -57,8 +58,9 @@ export function createCodexStructuredItemStreams( const forgetState = (key: string): void => { coalescer.forget(key) - states.delete(key) + states.forget(key) checkpointLengths.delete(key) + pendingCheckpoints.delete(key) const pending = pendingPatches.get(key) if (pending) { retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending)) @@ -67,8 +69,8 @@ export function createCodexStructuredItemStreams( } const trimStates = (): void => { - while (states.size > MAX_CODEX_ITEM_STREAM_STATES) { - const oldest = states.keys().next().value + while (states.overCapacity) { + const oldest = states.oldestEvictable() if (typeof oldest !== 'string') { break } @@ -124,6 +126,7 @@ export function createCodexStructuredItemStreams( const state = states.get(key) if (state && append(state, text)) { checkpointLengths.set(key, text.length) + pendingCheckpoints.delete(key) return true } return false @@ -133,6 +136,7 @@ export function createCodexStructuredItemStreams( windowMs: deps.coalesceMs, maxRetainedBytes: deps.maxRetainedBytes, maxTotalRetainedBytes: deps.maxTotalRetainedBytes, + isProtected: (key) => states.isPersistent(key), schedule: deps.schedule, emit: (key, text) => { return persist(key, text, false) @@ -144,7 +148,7 @@ export function createCodexStructuredItemStreams( itemId: string, type: string, params: unknown - ): CodexItemStreamState => { + ): CodexItemStreamState | null => { const key = codexStructuredItemKey(threadId, itemId) const existing = states.get(key) if (existing) { @@ -152,36 +156,25 @@ export function createCodexStructuredItemStreams( } const item = { type, id: itemId } const state = { item, identity: deps.identityFor(threadId, params, item) } - states.set(key, state) + if (!states.retain(key, state)) { + return null + } trimStates() return state } const flush = (): boolean => { let flushed = coalescer.flushAll() - for (const key of states.keys()) { + for (const key of pendingCheckpoints) { const snapshot = coalescer.snapshot(key) if (snapshot && checkpointLengths.get(key) !== snapshot.text.length) { flushed = persist(key, snapshot.text, true) && flushed + } else { + pendingCheckpoints.delete(key) } } - for (const [key, pending] of pendingPatches) { - const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(pending.identity, pending.body) - : (deps.sink.appendItem(pending.identity, pending.body), { accepted: true as const }) - if (!admission.accepted) { - flushed = false - continue - } - const published = deps.sink.tryPublish - ? deps.sink.tryPublish() - : (deps.sink.publish(), { accepted: true as const }) - if (!published.accepted) { - flushed = false - continue - } - retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending)) - pendingPatches.delete(key) + for (const key of pendingPatches.keys()) { + flushed = flushPatch(key).accepted && flushed } return flushed } @@ -209,11 +202,21 @@ export function createCodexStructuredItemStreams( } return { + get persistentCount() { + return states.persistentSize + }, + canTrack: (threadId, item, identity) => + states.canRetain(codexStructuredItemKey(threadId, item.id), { + item: boundStreamItem(item) as CodexThreadItem, + identity + }), track: (threadId, item, identity) => { const key = codexStructuredItemKey(threadId, item.id) - states.delete(key) - states.set(key, { item: boundStreamItem(item) as CodexThreadItem, identity }) + if (!states.retain(key, { item: boundStreamItem(item) as CodexThreadItem, identity })) { + return false + } trimStates() + return true }, handle: (threadId, method, params) => { const paramsRecord = readCodexItemStreamRecord(params) @@ -230,6 +233,9 @@ export function createCodexStructuredItemStreams( const key = codexStructuredItemKey(threadId, itemId) const streamFlushed = coalescer.flush(key) const state = ensureState(threadId, itemId, 'fileChange', params) + if (!state) { + return { handled: true, admission: { accepted: false, reason: 'failed' } } + } state.item = { ...state.item, changes: paramsRecord.changes } const translated = codexJournalItem(state.item) if (translated.body) { @@ -269,9 +275,14 @@ export function createCodexStructuredItemStreams( return { handled: true, admission: { accepted: true } } } const state = ensureState(threadId, itemId, type ?? 'reasoning', params) + if (!state) { + return { handled: true, admission: { accepted: false, reason: 'failed' } } + } const delta = method === REASONING_PART_METHOD ? '\n' : paramsRecord.delta if (typeof delta === 'string') { - const accepted = coalescer.append(codexStructuredItemKey(threadId, state.item.id), delta) + const key = codexStructuredItemKey(threadId, state.item.id) + pendingCheckpoints.add(key) + const accepted = coalescer.append(key, delta) if (!accepted) { return { handled: true, admission: { accepted: false, reason: 'backpressure' } } } @@ -287,6 +298,7 @@ export function createCodexStructuredItemStreams( coalescer.dispose() states.clear() checkpointLengths.clear() + pendingCheckpoints.clear() pendingPatches.clear() retainedPatchBytes = 0 }, diff --git a/src/main/codex/codex-structured-item-translation.test.ts b/src/main/codex/codex-structured-item-translation.test.ts index 45afb0c9fa5..201df58bc90 100644 --- a/src/main/codex/codex-structured-item-translation.test.ts +++ b/src/main/codex/codex-structured-item-translation.test.ts @@ -584,6 +584,16 @@ describe('codex item bodies', () => { }) }) + it('preserves plan prose documents byte-for-byte as status text', () => { + const text = + ' # Implementation plan\r\n\r\n- [ ] Preserve prose\r\n- [x] Keep café → 日本語\r\n\r\n```ts\r\nconst task = "pending"\r\n```\r\n ' + + expect(codexJournalItem({ type: 'plan', id: 'plan-document', text })).toEqual({ + body: { kind: 'status', text, presentation: 'plan-document' }, + handled: true + }) + }) + it('renders reasoning as status and exposes an unknown item as a provider frame', () => { expect(codexItemBody({ type: 'reasoning', id: 'r', text: 'thinking' })).toEqual({ kind: 'status', diff --git a/src/main/codex/codex-structured-journal-contracts.ts b/src/main/codex/codex-structured-journal-contracts.ts index acd04a4cf30..56a543fe001 100644 --- a/src/main/codex/codex-structured-journal-contracts.ts +++ b/src/main/codex/codex-structured-journal-contracts.ts @@ -1,11 +1,13 @@ import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' +import type { CodexSubagentExecutions } from './codex-subagent-executions' export type CodexJournalTranslatorDeps = { sink: StructuredAgentSessionEventSink bindPromptItemId?: (journalItemId: string, threadId: string, promptKey: string) => void primaryThreadId?: () => string | null + subagentExecutions?: CodexSubagentExecutions coalesceMs?: number maxRetainedBytes?: number schedule?: AgentSessionDeltaCoalescerDeps['schedule'] diff --git a/src/main/codex/codex-structured-journal-items.ts b/src/main/codex/codex-structured-journal-items.ts index f984bc1d9bf..62091580da3 100644 --- a/src/main/codex/codex-structured-journal-items.ts +++ b/src/main/codex/codex-structured-journal-items.ts @@ -11,7 +11,8 @@ import { type CodexThreadItem } from './codex-structured-item-translation' import { createCodexStructuredItemStreams } from './codex-structured-item-streams' -import { codexStructuredItemKey } from './codex-structured-item-stream-bounds' +import { boundStreamItem, codexStructuredItemKey } from './codex-structured-item-stream-bounds' +import { codexCommandOutlivesTurn } from './codex-command-lifecycle' import type { CodexItemTranslation, CodexJournalTranslationAdmission, @@ -40,7 +41,7 @@ export class CodexJournalItems { private readonly deps: Pick< CodexJournalTranslatorDeps, 'sink' | 'coalesceMs' | 'maxRetainedBytes' | 'schedule' - >, + > & { maxMetadataBytes?: number }, private readonly activeTurn: (threadId: string) => string | null, private readonly suppress: (threadId: string, turnId: string) => void ) { @@ -49,6 +50,7 @@ export class CodexJournalItems { coalesceMs: deps.coalesceMs, maxRetainedBytes: deps.maxRetainedBytes, schedule: deps.schedule, + maxMetadataBytes: deps.maxMetadataBytes, identityFor: (threadId, params, item) => { const turnId = readCodexTurnId(params) ?? this.activeTurn(threadId) return this.identityFor(threadId, turnId, item) @@ -81,6 +83,12 @@ export class CodexJournalItems { if (item.type === 'contextCompaction' && event.method === 'item/started') { return { handled: true, admission: CODEX_JOURNAL_ADMITTED } } + if ( + event.method !== 'item/completed' && + !this.streams.canTrack(event.threadId, item, identity) + ) { + return { handled: true, admission: { accepted: false, reason: 'failed' } } + } const translated = codexJournalItem(item) const command = readCodexJournalString(item, 'command') if (command) { @@ -157,12 +165,15 @@ export class CodexJournalItems { item: CodexThreadItem, identity: AgentJournalItemIdentity ): void { - this.streams.track(threadId, item, identity) + const retainedItem = codexCommandOutlivesTurn(item) + ? (boundStreamItem(item) as CodexThreadItem) + : item + this.streams.track(threadId, retainedItem, identity) this.activeItems.set(codexStructuredItemKey(threadId, item.id), { threadId, turnId, identity, - item + item: retainedItem }) } @@ -194,8 +205,10 @@ export class CodexJournalItems { } private trimActiveState(): CodexJournalTranslationAdmission { - while (this.activeItems.size > MAX_CODEX_ACTIVE_ITEMS) { - const oldest = this.activeItems.keys().next().value + while (this.activeItems.size - this.streams.persistentCount > MAX_CODEX_ACTIVE_ITEMS) { + const oldest = [...this.activeItems].find( + ([, active]) => !codexCommandOutlivesTurn(active.item) + )?.[0] if (typeof oldest !== 'string') { break } diff --git a/src/main/codex/codex-structured-journal-settlement.ts b/src/main/codex/codex-structured-journal-settlement.ts index 2b8eb627d4f..8c8bf091839 100644 --- a/src/main/codex/codex-structured-journal-settlement.ts +++ b/src/main/codex/codex-structured-journal-settlement.ts @@ -20,6 +20,7 @@ import { } from './codex-structured-item-translation' import type { CodexStructuredItemStreams } from './codex-structured-item-streams' import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' +import { codexCommandOutlivesTurn } from './codex-command-lifecycle' export type CodexActiveJournalItem = { threadId: string @@ -118,6 +119,9 @@ export function settleCodexJournalTurn(input: { if (active.threadId !== input.threadId || active.turnId !== input.turnId) { continue } + if (codexCommandOutlivesTurn(active.item)) { + continue + } const streamed = input.streams.snapshot(active.threadId, active.item.id) const translated = streamed ? codexStreamingJournalItem(active.item, streamed.text) diff --git a/src/main/codex/codex-structured-journal-translation-frames.ts b/src/main/codex/codex-structured-journal-translation-frames.ts index 22dc516b210..d4cb2c825d9 100644 --- a/src/main/codex/codex-structured-journal-translation-frames.ts +++ b/src/main/codex/codex-structured-journal-translation-frames.ts @@ -6,6 +6,8 @@ * than as the shape checks each arm performs. */ +import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' +import type { CodexJournalItems } from './codex-structured-journal-items' import type { CodexJournalTranslationAdmission } from './codex-structured-journal-contracts' import { settleCodexOversizedNotification } from './codex-structured-journal-settlement' import { @@ -41,3 +43,23 @@ export function settleCodexOversizedNotificationFrame(input: { }) : null } + +export function createCodexOversizedNotificationSettler( + deps: { sink: OversizedInput['sink'] }, + items: Pick +) { + return settleOversizedNotification + + /** Settles the item a notification the transport refused to carry left + * mid-flight; null when the frame is not one. */ + function settleOversizedNotification( + event: Extract + ): CodexJournalTranslationAdmission | null { + return settleCodexOversizedNotificationFrame({ + ...event, + sink: deps.sink, + streams: items.streams, + activeItems: items.activeItems + }) + } +} diff --git a/src/main/codex/codex-structured-journal-translation-subagents.test.ts b/src/main/codex/codex-structured-journal-translation-subagents.test.ts index bf5cdffa5a9..8db30d31b21 100644 --- a/src/main/codex/codex-structured-journal-translation-subagents.test.ts +++ b/src/main/codex/codex-structured-journal-translation-subagents.test.ts @@ -59,6 +59,22 @@ function deliverActivity( translator: ReturnType, params: unknown ): void { + const item = (params as { item: { kind: string; agentThreadId: string } }).item + if (item.kind === 'started' || item.kind === 'completed') { + translator.handle({ + type: 'notification', + sessionId: SESSION_ID, + threadId: item.agentThreadId, + method: item.kind === 'started' ? 'turn/started' : 'turn/completed', + params: { + threadId: item.agentThreadId, + turn: { + id: `execution:${item.agentThreadId}`, + status: item.kind === 'started' ? 'inProgress' : 'completed' + } + } + }) + } translator.handle(notification('item/started', params)) translator.handle(notification('item/completed', params)) } diff --git a/src/main/codex/codex-structured-journal-translation.ts b/src/main/codex/codex-structured-journal-translation.ts index 00b90d7ffc8..cea129e5cdf 100644 --- a/src/main/codex/codex-structured-journal-translation.ts +++ b/src/main/codex/codex-structured-journal-translation.ts @@ -19,7 +19,7 @@ import { settleCodexJournalSession, settleCodexJournalTurn } from './codex-structured-journal-settlement' -import { settleCodexOversizedNotificationFrame } from './codex-structured-journal-translation-frames' +import { createCodexOversizedNotificationSettler } from './codex-structured-journal-translation-frames' import { restoreCodexJournalThread } from './codex-structured-journal-translation-restore' import { CodexJournalActiveTurns } from './codex-structured-journal-translation-turn-state' import { publishCodexTurnLifecycle } from './codex-structured-journal-translation-turns' @@ -58,13 +58,15 @@ export function createCodexJournalTranslator( (threadId) => activeTurns.current(threadId), (threadId, turnId) => genericFrames.suppress(threadId, turnId) ) + const settleOversizedNotification = createCodexOversizedNotificationSettler(deps, items) const prompts = new CodexJournalPrompts(deps, (threadId, itemId) => items.detailFor(threadId, itemId) ) const subagents = new CodexSubagentRoster({ sink: deps.sink, primaryThreadId: () => deps.primaryThreadId?.() ?? null, - activeTurn: (threadId) => activeTurns.current(threadId) + activeTurn: (threadId) => activeTurns.current(threadId), + ...(deps.subagentExecutions ? { executions: deps.subagentExecutions } : {}) }) const flushStreams = (): CodexJournalTranslationAdmission => items.streams.flush() ? CODEX_JOURNAL_ADMITTED : { accepted: false, reason: 'backpressure' } @@ -174,16 +176,17 @@ export function createCodexJournalTranslator( } return genericFrames.appendUnhandled(event.kind, event.payload, event.threadId) } - if (event.method === 'turn/started') { - return startTurn(event) + if (event.method === 'turn/started' || event.method === 'turn/completed') { + const childAdmission = subagents.handleTurnEvent(event) + if (!childAdmission.accepted) { + return childAdmission + } + return event.method === 'turn/started' ? startTurn(event) : completeTurn(event) } const compaction = compactions.handle(event) if (compaction) { return publishActivity(event, compaction) } - if (event.method === 'turn/completed') { - return completeTurn(event) - } if (event.method === CODEX_TOKEN_USAGE_METHOD) { // Classified `status-chrome`, so the generic-frame path swallows it // before the journal. The roster consumes it as a typed notification. @@ -240,19 +243,6 @@ export function createCodexJournalTranslator( } } - /** Settles the item a notification the transport refused to carry left - * mid-flight; null when the frame is not one. */ - function settleOversizedNotification( - event: Extract - ): CodexJournalTranslationAdmission | null { - return settleCodexOversizedNotificationFrame({ - ...event, - sink: deps.sink, - streams: items.streams, - activeItems: items.activeItems - }) - } - function startTurn( event: Extract ): CodexJournalTranslationAdmission { diff --git a/src/main/codex/codex-structured-session-acquire.ts b/src/main/codex/codex-structured-session-acquire.ts index 8c8b39ca48b..3f7f130e173 100644 --- a/src/main/codex/codex-structured-session-acquire.ts +++ b/src/main/codex/codex-structured-session-acquire.ts @@ -8,6 +8,8 @@ import { closeFailedCodexAcquisition, stopSupersededCodexAcquisition } from './codex-structured-acquisition-lifecycle' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' +import { CodexSubagentExecutions } from './codex-subagent-executions' import { createCodexJournalTranslator } from './codex-structured-journal-translation' import { openCodexAppServerConnection } from './codex-app-server-connection' import { codexProcessIdentity, codexProviderHandleLink } from './codex-structured-owner-identity' @@ -74,10 +76,12 @@ export async function acquireCodexStructuredSession(input: { acquireInput.identity.providerHandle.kind === 'codex' ? acquireInput.identity.providerHandle.threadId : null + const subagentExecutions = new CodexSubagentExecutions() const translator = acquireInput.events ? createCodexJournalTranslator({ sink: acquireInput.events, primaryThreadId: () => primaryThreadId, + subagentExecutions, bindPromptItemId: (journalItemId, threadId, promptKey) => acquisition.prompts.bindJournalItemId(journalItemId, threadId, promptKey) }) @@ -138,6 +142,7 @@ export async function acquireCodexStructuredSession(input: { connection: acquisition.connection, error, prompts: acquisition.prompts, + onBackgroundTasksChanged: deps.onBackgroundTasksChanged, ...(deps.onEvent ? { onEvent: deps.onEvent } : {}) }) } finally { @@ -199,6 +204,7 @@ export async function acquireCodexStructuredSession(input: { reportedOptions: reportedCodexThreadOptions(opened), turnIdWaiters: [], translator, + backgroundTasks: new CodexBackgroundTaskTracker(opened.threadId, subagentExecutions), forceCloseUnexpected: (reason) => input.forceCloseUnexpected( sessionId, diff --git a/src/main/codex/codex-structured-session-adapter.ts b/src/main/codex/codex-structured-session-adapter.ts index ebd7c3331bf..8b0e649b177 100644 --- a/src/main/codex/codex-structured-session-adapter.ts +++ b/src/main/codex/codex-structured-session-adapter.ts @@ -16,11 +16,7 @@ import type { CodexJournalTranslationAdmission } from './codex-structured-journa import { answerCodexPrompt } from './codex-structured-prompt-replies' import { dispatchCodexTurn, isCodexTurnOptionKey } from './codex-structured-turn-start' import { supportsCodexStructuredLocation } from './codex-structured-location-support' -import { - closeAllCodexSessions, - closeCodexPublishedSession, - closeCodexSession -} from './codex-structured-session-close' +import { CodexStructuredSessionTeardown } from './codex-structured-session-teardown' import { applyCodexStructuredSessionOption, readLiveCodexSessionOptions @@ -54,6 +50,7 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap private readonly acquisitions = new CodexAcquisitionRegistry() private readonly turnCancellation: CodexStructuredTurnCancellation private readonly notificationRetries: ReturnType + private readonly teardown: CodexStructuredSessionTeardown constructor(private readonly deps: CodexStructuredSessionAdapterDeps) { this.notificationRetries = createCodexStructuredNotificationRetry({ @@ -61,6 +58,15 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap translate: (sessionId, session, method, params) => this.translateNotification(sessionId, session, method, params) }) + this.teardown = new CodexStructuredSessionTeardown({ + sessions: this.sessions, + acquisitions: this.acquisitions, + ...(deps.onEvent ? { onEvent: deps.onEvent } : {}), + ...(deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: deps.onBackgroundTasksChanged } + : {}), + forgetNotificationRetries: (sessionId) => this.notificationRetries.clear(sessionId, null) + }) this.turnCancellation = new CodexStructuredTurnCancellation({ captureTurnProcesses: deps.captureTurnProcesses, terminateTurnProcesses: deps.terminateTurnProcesses, @@ -92,7 +98,7 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap handleUnhandledFrame: (sessionId, kind, payload) => this.handleUnhandledFrame(sessionId, kind, payload), forceCloseUnexpected: (sessionId, fence, acquisitionGeneration, reason) => - this.forceCloseUnexpected(sessionId, fence, acquisitionGeneration, reason) + this.teardown.forceCloseUnexpected(sessionId, fence, acquisitionGeneration, reason) }) /** Buffers pre-publication events and drops events from superseded children. */ @@ -134,12 +140,20 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap session: CodexSession, event: CodexStructuredSessionEvent ): CodexJournalTranslationAdmission { + if (event.type === 'notification' && !session.backgroundTasks.canObserve(event)) { + return { accepted: false, reason: 'failed' } + } const admission = session.translator?.handle(event) ?? { accepted: true } if (!admission.accepted) { return admission } if (event.type === 'notification') { this.compactions.codex(event.sessionId, event.method, event.params) + // After the admission check, so a refused frame is observed by the strip + // only on the retry that also reaches the journal. + if (session.backgroundTasks.observe(event)) { + this.deps.onBackgroundTasksChanged?.(event.sessionId, session.backgroundTasks.state) + } } if (event.type === 'ended') { this.compactions.ended(event.sessionId) @@ -167,6 +181,10 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap ) } + backgroundTaskState: NonNullable = ( + sessionId + ) => this.sessions.get(sessionId)?.backgroundTasks.state + bindPromptItemId = (sessionId: string, journalItemId: string, promptKey: string): void => this.sessions .get(sessionId) @@ -267,59 +285,12 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap identity: AgentSessionJournalIdentity }): Promise => this.sessions.get(input.identity.sessionId)?.historyPath ?? null - closeSession = async (sessionId: string): Promise => { - const closed = await closeCodexSession( - sessionId, - this.sessions, - this.acquisitions, - this.deps.onEvent - ) - if (closed) { - this.notificationRetries.clear(sessionId, null) - } - return closed - } - forceCloseSession = async (sessionId: string): Promise => { - const closed = await closeCodexPublishedSession(this.sessions, sessionId, this.deps.onEvent, { - allowFailedSettlement: true, - requestedClose: false - }) - if (closed) { - this.notificationRetries.clear(sessionId, null) - } - return closed - } - - private forceCloseUnexpected( - sessionId: string, - fence: number, - acquisitionGeneration: string, - reason: Error - ): Promise { - const session = this.sessions.get(sessionId) - if ( - !session || - session.ended || - session.fence !== fence || - session.acquisitionGeneration !== acquisitionGeneration - ) { - return Promise.resolve(false) - } - return closeCodexPublishedSession(this.sessions, sessionId, this.deps.onEvent, { - allowFailedSettlement: true, - requestedClose: false, - expectedFence: fence, - expectedAcquisitionGeneration: acquisitionGeneration, - unexpectedReason: reason - }) - } - disposeSession = (sessionId: string): Promise => this.closeSession(sessionId) - closeAll = (): Promise => - closeAllCodexSessions(this.sessions, this.acquisitions, (sessionId) => - this.disposeSession(sessionId) - ) + closeSession = (sessionId: string): Promise => this.teardown.close(sessionId) + forceCloseSession = (sessionId: string): Promise => this.teardown.forceClose(sessionId) + disposeSession = (sessionId: string): Promise => this.teardown.close(sessionId) + closeAll = (): Promise => this.teardown.closeAll() releaseAcquisition = (input: { sessionId: string }): Promise => - this.closeSession(input.sessionId) + this.teardown.close(input.sessionId) private session(sessionId: string): CodexSession { return requireLiveCodexSession(this.sessions, sessionId) diff --git a/src/main/codex/codex-structured-session-background-tasks.test.ts b/src/main/codex/codex-structured-session-background-tasks.test.ts new file mode 100644 index 00000000000..aa6156027e8 --- /dev/null +++ b/src/main/codex/codex-structured-session-background-tasks.test.ts @@ -0,0 +1,258 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import type { + CodexAppServerConnection, + CodexAppServerConnectionHandlers, + openCodexAppServerConnection +} from './codex-app-server-connection' +import { CodexStructuredSessionAdapter } from './codex-structured-session-adapter' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' +import type { CodexStructuredSessionEvent } from './codex-structured-session-state' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' + +// Proves the strip is actually REACHED from provider traffic: the tracker is +// unit-tested separately, and a producer that is correct but unwired publishes +// nothing while every one of its own tests stays green. + +const THREAD_ID = '01a07d54-3785-71d0-b065-82c8ebbc572a' +const PARENT_TURN = '01a07d54-37be-72e1-8206-8f0c23dd2cef' +const CHILD_ID = '01a07d54-5523-78a3-91f5-e0acb1dab065' + +/** A three-route stand-in, deliberately smaller than the full adapter harness: + * this suite only needs a thread and a notification pipe. */ +function fakeCodex(close: () => Promise = async () => true): { + handlers: () => CodexAppServerConnectionHandlers + openConnection: typeof openCodexAppServerConnection +} { + let live: CodexAppServerConnectionHandlers = {} + const openConnection = (async (_launch, handlers = {}) => { + live = handlers + const connection: CodexAppServerConnection = { + pid: 4321, + closed: false, + request: async (method) => + method === 'thread/start' ? { thread: { id: THREAD_ID, path: null } } : {}, + notify: () => {}, + respond: () => {}, + respondWithError: () => {}, + close + } as unknown as CodexAppServerConnection + return connection + }) as typeof openCodexAppServerConnection + return { handlers: () => live, openConnection } +} + +function identity(sessionId: string): AgentSessionJournalIdentity { + return { + sessionId, + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD_ID } + } +} + +function subagentNotification(kind: string): { method: string; params: unknown } { + return { + method: 'item/started', + params: { + item: { + type: 'subAgentActivity', + id: 'call_1', + kind, + agentThreadId: CHILD_ID, + agentPath: '/root/count_a' + }, + threadId: THREAD_ID, + turnId: PARENT_TURN + } + } +} + +const TURN_COMPLETED = { + method: 'turn/completed', + params: { threadId: THREAD_ID, turn: { id: PARENT_TURN, status: 'completed' } } +} + +async function adapterWithSession( + published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[], + events?: StructuredAgentSessionEventSink, + onEvent?: (event: CodexStructuredSessionEvent) => void, + close?: () => Promise +): Promise<{ adapter: CodexStructuredSessionAdapter; codex: ReturnType }> { + const codex = fakeCodex(close) + const adapter = new CodexStructuredSessionAdapter({ + resolveLaunch: async () => ({ + command: 'codex', + args: ['app-server'], + cwd: '/work/repo', + codexHome: null, + resumeThreadId: null + }), + openConnection: codex.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + onEvent, + onBackgroundTasksChanged: (sessionId, state) => published.push({ sessionId, state }) + }) + await adapter.acquire({ + identity: identity('session-1'), + fence: 7, + spawnToken: 'spawn-9', + events + }) + codex.handlers().onNotification?.('turn/started', { + threadId: THREAD_ID, + turn: { id: PARENT_TURN, status: 'inProgress' } + }) + codex.handlers().onNotification?.('turn/started', { + threadId: CHILD_ID, + turn: { id: 'child-turn', status: 'inProgress' } + }) + return { adapter, codex } +} + +describe('codex background tasks reach the strip', () => { + it('clears natural-exit state before lifecycle observers can read it', async () => { + const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] + const onEvent = vi.fn() + const { adapter, codex } = await adapterWithSession(published, undefined, onEvent) + const spawn = subagentNotification('started') + codex.handlers().onNotification?.(spawn.method, spawn.params) + codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params) + expect(adapter.backgroundTaskState('session-1')?.tasks).toHaveLength(1) + published.length = 0 + onEvent.mockImplementation((event: CodexStructuredSessionEvent) => { + if (event.type === 'ended') { + expect(adapter.backgroundTaskState('session-1')).toBeNull() + } + }) + codex.handlers().onExit?.(new Error('provider exited')) + expect(adapter.backgroundTaskState('session-1')).toBeNull() + expect(published).toEqual([{ sessionId: 'session-1', state: null }]) + await adapter.closeSession('session-1') + }) + + it('keeps live tasks when close is refused', async () => { + const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] + const close = vi.fn(async () => false) + const { adapter, codex } = await adapterWithSession(published, undefined, undefined, close) + const spawn = subagentNotification('started') + codex.handlers().onNotification?.(spawn.method, spawn.params) + codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params) + const before = adapter.backgroundTaskState('session-1') + published.length = 0 + expect(await adapter.closeSession('session-1')).toBe(false) + expect(adapter.backgroundTaskState('session-1')).toEqual(before) + expect(published).toEqual([]) + close.mockResolvedValue(true) + await adapter.closeSession('session-1') + }) + + it('does not let an old exit callback clear a replacement roster', async () => { + const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] + const { adapter, codex } = await adapterWithSession(published) + const oldExit = codex.handlers().onExit + await adapter.acquire({ identity: identity('session-1'), fence: 8, spawnToken: 'spawn-10' }) + codex.handlers().onNotification?.('turn/started', { + threadId: CHILD_ID, + turn: { id: 'replacement-child-turn' } + }) + const spawn = subagentNotification('started') + codex.handlers().onNotification?.(spawn.method, spawn.params) + const before = adapter.backgroundTaskState('session-1') + expect(before?.tasks).toHaveLength(1) + published.length = 0 + oldExit?.(new Error('old provider exited late')) + expect(adapter.backgroundTaskState('session-1')).toEqual(before) + expect(published).toEqual([]) + await adapter.closeSession('session-1') + }) + + it('recovers the exact provider generation when command metadata cannot be admitted', async () => { + const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] + const observed: CodexStructuredSessionEvent[] = [] + const appendItem = vi.fn() + const { adapter, codex } = await adapterWithSession( + published, + { appendItem, appendTombstone: () => {}, publish: () => {} }, + (event) => observed.push(event) + ) + appendItem.mockClear() + observed.length = 0 + const admission = vi + .spyOn(CodexBackgroundTaskTracker.prototype, 'canObserve') + .mockReturnValue(false) + try { + codex.handlers().onNotification?.('item/started', { + threadId: THREAD_ID, + turnId: PARENT_TURN, + item: { + type: 'commandExecution', + id: 'over-budget', + command: 'sleep 1', + source: 'unifiedExecStartup', + status: 'inProgress' + } + }) + await vi.waitFor(() => expect(adapter.backgroundTaskState('session-1')).toBeUndefined()) + expect(appendItem.mock.calls.map((call) => call[1])).toEqual([ + { kind: 'status', text: 'Provider exited: notification admission failed (failed)' } + ]) + expect(observed).toEqual([ + expect.objectContaining({ + type: 'ended', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: expect.any(String), + reason: 'notification admission failed (failed)' + }) + ]) + expect(published).toEqual([{ sessionId: 'session-1', state: null }]) + } finally { + admission.mockRestore() + await adapter.closeSession('session-1') + } + }) + + it('publishes the orphaned fan-out once the spawning turn completes', async () => { + const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] + const { adapter, codex } = await adapterWithSession(published) + + const spawn = subagentNotification('started') + codex.handlers().onNotification?.(spawn.method, spawn.params) + // The child is still inside the turn, so the strip stays silent. + expect(published).toEqual([]) + expect(adapter.backgroundTaskState('session-1')).toBeNull() + + codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params) + + expect(published).toEqual([ + { + sessionId: 'session-1', + state: { + state: 'monitoring', + supportsStopAll: false, + tasks: [{ id: `codex-agent:${CHILD_ID}`, kind: 'agent', description: 'count_a' }] + } + } + ]) + expect(adapter.backgroundTaskState('session-1')).toEqual(published[0].state) + }) + + it('clears the strip when the session closes', async () => { + const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] + const { adapter, codex } = await adapterWithSession(published) + const spawn = subagentNotification('started') + codex.handlers().onNotification?.(spawn.method, spawn.params) + codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params) + published.length = 0 + + expect(await adapter.closeSession('session-1')).toBe(true) + + // Explicit null, not silence: the reader answers `undefined` once the + // session is gone, which every channel treats as "unchanged". + expect(published).toEqual([{ sessionId: 'session-1', state: null }]) + expect(adapter.backgroundTaskState('session-1')).toBeUndefined() + }) +}) diff --git a/src/main/codex/codex-structured-session-close.test.ts b/src/main/codex/codex-structured-session-close.test.ts index b04e7bc2540..45bfbbf45a1 100644 --- a/src/main/codex/codex-structured-session-close.test.ts +++ b/src/main/codex/codex-structured-session-close.test.ts @@ -10,6 +10,7 @@ import { type CodexStructuredSessionEvent } from './codex-structured-session-adapter' import { handleCodexSessionExit } from './codex-structured-session-close' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' import type { CodexSession } from './codex-structured-session-state' import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter' import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' @@ -90,6 +91,7 @@ describe('Codex structured session close lifecycle', () => { } as unknown as NonNullable const session = { connection, + backgroundTasks: new CodexBackgroundTaskTracker('thread-1'), ended: false, requestedClose: false, fence: 7, diff --git a/src/main/codex/codex-structured-session-close.ts b/src/main/codex/codex-structured-session-close.ts index db723096177..5b29dc6c056 100644 --- a/src/main/codex/codex-structured-session-close.ts +++ b/src/main/codex/codex-structured-session-close.ts @@ -4,6 +4,7 @@ import { cancelCodexAcquisitionAttempt, type CodexAcquisitionRegistry, type CodexSession, + type CodexStructuredSessionAdapterDeps, type CodexStructuredSessionEvent } from './codex-structured-session-state' import type { StructuredAgentSessionLifecycleEvent } from '../native-chat/agent-session-wire/structured-agent-session-adapter' @@ -16,6 +17,7 @@ export function handleCodexSessionExit(input: { prompts?: CodexSession['prompts'] allowFailedSettlement?: boolean onEvent?: (event: CodexStructuredSessionEvent) => void + onBackgroundTasksChanged?: CodexStructuredSessionAdapterDeps['onBackgroundTasksChanged'] }): boolean { const session = input.sessions.get(input.sessionId) if (!session || session.connection !== input.connection || session.ended) { @@ -43,6 +45,8 @@ export function handleCodexSessionExit(input: { event.settlementRetryRequired = true } session.ended = true + session.backgroundTasks.clear() + input.onBackgroundTasksChanged?.(input.sessionId, null) session.unbindReadingControl?.() input.onEvent?.(event) session.prompts.clear() diff --git a/src/main/codex/codex-structured-session-options.test.ts b/src/main/codex/codex-structured-session-options.test.ts index 1f28ff5e197..b081e52dd6a 100644 --- a/src/main/codex/codex-structured-session-options.test.ts +++ b/src/main/codex/codex-structured-session-options.test.ts @@ -7,6 +7,7 @@ import { reportedCodexThreadOptions, restoredCodexSessionOptions } from './codex-structured-session-options' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' import type { CodexSession } from './codex-structured-session-state' function optionSession(request: CodexAppServerConnection['request']): CodexSession { @@ -20,6 +21,7 @@ function optionSession(request: CodexAppServerConnection['request']): CodexSessi respondWithError: () => {}, close: async () => true }, + backgroundTasks: new CodexBackgroundTaskTracker('thread-1'), ended: false, requestedClose: false, fence: 1, diff --git a/src/main/codex/codex-structured-session-state.ts b/src/main/codex/codex-structured-session-state.ts index 12ba28d712f..d004b0006e6 100644 --- a/src/main/codex/codex-structured-session-state.ts +++ b/src/main/codex/codex-structured-session-state.ts @@ -6,6 +6,8 @@ import type { openCodexAppServerConnection } from './codex-app-server-connection' import { CodexAcquisitionWindow } from './codex-structured-acquisition-window' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import type { CodexBackgroundTaskTracker } from './codex-background-task-tracker' import type { CodexJournalTranslator } from './codex-structured-journal-translation' import type { CodexTurnProcessSnapshot } from './codex-structured-turn-processes' import type { StructuredAgentSessionLifecycleEvent } from '../native-chat/agent-session-wire/structured-agent-session-adapter' @@ -44,6 +46,10 @@ export type CodexStructuredSessionAdapterDeps = { /** Host capability seam; production uses the native Windows process table. */ isWindowsProcessStartTimeAvailable?: () => boolean onEvent?: (event: CodexStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void openConnection?: typeof openCodexAppServerConnection readProcessStartTime?: (pid: number) => Promise mintLinkId?: () => string @@ -73,6 +79,8 @@ export type CodexSession = { reportedOptions: { model?: string; effort?: string } turnIdWaiters: ((turnId: string) => void)[] translator: CodexJournalTranslator | null + /** Ephemeral roster behind the background-tasks strip; never durable state. */ + backgroundTasks: CodexBackgroundTaskTracker unbindReadingControl?: () => void /** Terminates this exact child as an unexpected death and enters host recovery. */ forceCloseUnexpected?: (reason: Error) => Promise diff --git a/src/main/codex/codex-structured-session-teardown.ts b/src/main/codex/codex-structured-session-teardown.ts new file mode 100644 index 00000000000..dcd38eb82fe --- /dev/null +++ b/src/main/codex/codex-structured-session-teardown.ts @@ -0,0 +1,94 @@ +// Stopping one Codex app-server child, in the four ways the host asks for it. +// +// Every path funnels through `settled` so the ephemeral surfaces a closed +// session owns are cleared exactly once, and only when the child was actually +// proven stopped — a refused close leaves the session indexed for a retry. + +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import { + closeAllCodexSessions, + closeCodexPublishedSession, + closeCodexSession +} from './codex-structured-session-close' +import type { + CodexAcquisitionRegistry, + CodexSession, + CodexStructuredSessionEvent +} from './codex-structured-session-state' + +export type CodexStructuredSessionTeardownDeps = { + sessions: Map + acquisitions: CodexAcquisitionRegistry + onEvent?: (event: CodexStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void + forgetNotificationRetries: (sessionId: string) => void +} + +export class CodexStructuredSessionTeardown { + constructor(private readonly deps: CodexStructuredSessionTeardownDeps) {} + + close = async (sessionId: string): Promise => { + const closed = await closeCodexSession( + sessionId, + this.deps.sessions, + this.deps.acquisitions, + this.deps.onEvent + ) + return this.settled(sessionId, closed) + } + + forceClose = async (sessionId: string): Promise => { + const closed = await closeCodexPublishedSession( + this.deps.sessions, + sessionId, + this.deps.onEvent, + { allowFailedSettlement: true, requestedClose: false } + ) + return this.settled(sessionId, closed) + } + + /** Terminates this exact child as an unexpected death. Every ownership check + * stays here so a stale caller cannot close a replacement child. */ + forceCloseUnexpected = ( + sessionId: string, + fence: number, + acquisitionGeneration: string, + reason: Error + ): Promise => { + const session = this.deps.sessions.get(sessionId) + if ( + !session || + session.ended || + session.fence !== fence || + session.acquisitionGeneration !== acquisitionGeneration + ) { + return Promise.resolve(false) + } + return closeCodexPublishedSession(this.deps.sessions, sessionId, this.deps.onEvent, { + allowFailedSettlement: true, + requestedClose: false, + expectedFence: fence, + expectedAcquisitionGeneration: acquisitionGeneration, + unexpectedReason: reason + }).then((closed) => this.settled(sessionId, closed)) + } + + closeAll = (): Promise => + closeAllCodexSessions(this.deps.sessions, this.deps.acquisitions, (sessionId) => + this.close(sessionId) + ) + + private settled(sessionId: string, closed: boolean): boolean { + if (closed) { + this.deps.forgetNotificationRetries(sessionId) + // Explicit null, not silence: the state reader answers `undefined` once + // the session leaves the map, which every channel reads as "unchanged" + // and would leave the last roster on screen. + this.deps.onBackgroundTasksChanged?.(sessionId, null) + } + return closed + } +} diff --git a/src/main/codex/codex-subagent-activity.ts b/src/main/codex/codex-subagent-activity.ts index f12e9dfb1b3..9761f8634f8 100644 --- a/src/main/codex/codex-subagent-activity.ts +++ b/src/main/codex/codex-subagent-activity.ts @@ -7,11 +7,10 @@ // segment is a semantic task name and the only label available. There is no // `thread/started` for a child, so nickname/role/depth do not exist. // * `agentsStates` on `collabAgentToolCall` arrived empty (`{}`) throughout the -// probe, so nothing here reads it — state comes from `kind` alone. +// probe, so nothing here reads it; child turn events own execution state. // * `thread/tokenUsage/updated` reports a per-thread RUNNING TOTAL, so the // latest frame replaces the previous one — it is never accumulated. -import type { NativeChatSubagentState } from '../../shared/native-chat-types' import type { CodexThreadItem } from './codex-structured-item-translation' export const CODEX_SUBAGENT_ITEM_TYPE = 'subAgentActivity' @@ -48,24 +47,6 @@ export function readCodexSubagentActivity(item: CodexThreadItem): CodexSubagentA } } -/** - * The state a `kind` implies for the child it names. - * - * An unrecognized kind means "this child exists and reported something we - * cannot classify" — `working`, which the session sweep will later settle to - * `unverifiable` if nothing better ever arrives. Claiming a terminal state from - * an unknown kind would assert an outcome the wire never gave us. - */ -export function codexSubagentStateForKind(kind: string): NativeChatSubagentState { - if (kind === 'completed') { - return 'completed' - } - if (kind === 'interrupted') { - return 'stopped' - } - return 'working' -} - /** Path segments, empty ones dropped: `/root/list_directory` → 2 segments. */ export function codexSubagentPathSegments(agentPath: string | null): string[] { return agentPath === null ? [] : agentPath.split('/').filter((part) => part.length > 0) diff --git a/src/main/codex/codex-subagent-execution-projection.test.ts b/src/main/codex/codex-subagent-execution-projection.test.ts new file mode 100644 index 00000000000..21d39f4f195 --- /dev/null +++ b/src/main/codex/codex-subagent-execution-projection.test.ts @@ -0,0 +1,211 @@ +import { describe, expect, it } from 'vitest' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' +import { createCodexJournalTranslator } from './codex-structured-journal-translation' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' +import { CodexSubagentExecutions } from './codex-subagent-executions' + +const PRIMARY = 'primary' +const CHILD = 'child' + +function turn( + method: 'turn/started' | 'turn/completed', + threadId: string, + turnId: string, + status = 'completed' +): CodexStructuredSessionEvent { + return { + type: 'notification', + sessionId: 'session', + method, + threadId, + params: { threadId, turn: { id: turnId, status } } + } +} + +function activity(parentTurn: string, kind = 'started'): CodexStructuredSessionEvent { + return { + type: 'notification', + sessionId: 'session', + threadId: PRIMARY, + method: 'item/started', + params: { + threadId: PRIMARY, + turnId: parentTurn, + item: { + type: 'subAgentActivity', + id: `activity-${parentTurn}-${kind}`, + kind, + agentThreadId: CHILD, + agentPath: '/root/task' + } + } + } +} + +function harness() { + const executions = new CodexSubagentExecutions() + const tracker = new CodexBackgroundTaskTracker(PRIMARY, executions) + const rows = new Map() + let refused = false + const translator = createCodexJournalTranslator({ + primaryThreadId: () => PRIMARY, + subagentExecutions: executions, + sink: { + appendItem: (identity, body) => rows.set(JSON.stringify(identity), body), + appendTombstone: () => {}, + publish: () => {}, + tryAppendItem: (identity, body) => { + if (refused) { + return { accepted: false, reason: 'backpressure' } + } + rows.set(JSON.stringify(identity), body) + return { accepted: true } + } + }, + schedule: (run) => { + run() + return () => {} + } + }) + function send(event: CodexStructuredSessionEvent) { + const admission = translator.handle(event) + if (admission.accepted && event.type === 'notification') { + tracker.observe(event) + } + return admission + } + function state(parentTurn: string): string | undefined { + for (const body of rows.values()) { + if (body.kind !== 'message') { + continue + } + for (const block of body.blocks) { + if (block.type === 'subagent-group' && block.groupId === `${PRIMARY}:${parentTurn}`) { + return block.agents[0]?.state + } + } + } + return undefined + } + return { + send, + tracker, + state, + refuse: (value: boolean) => { + refused = value + }, + dispose: () => translator.dispose() + } +} + +function firstRun(h: ReturnType) { + h.send(turn('turn/started', PRIMARY, 'parent-1')) + h.send(turn('turn/started', CHILD, 'child-1')) + h.send(activity('parent-1')) + h.send(turn('turn/completed', CHILD, 'child-1')) +} + +describe('shared child execution projection', () => { + it('creates no execution record from activity without an owner turn', () => { + const h = harness() + h.send(activity('parent-1')) + h.send(activity('parent-2', 'interacted')) + expect(h.state('parent-1')).toBeUndefined() + expect(h.state('parent-2')).toBeUndefined() + expect(h.tracker.state).toBeNull() + h.dispose() + }) + + it.each(['parent-1', 'parent-2'])( + 'reopens a child for real follow-up in %s and fences old execution events', + (parent) => { + const h = harness() + firstRun(h) + if (parent === 'parent-2') { + h.send(turn('turn/completed', PRIMARY, 'parent-1')) + h.send(turn('turn/started', PRIMARY, parent)) + } + h.send(activity(parent, 'interacted')) + expect(h.tracker.state).toBeNull() + h.send(turn('turn/started', CHILD, 'child-2')) + h.send(turn('turn/completed', PRIMARY, parent)) + expect(h.state(parent)).toBe('working') + expect(h.tracker.state?.tasks).toHaveLength(1) + h.send(turn('turn/completed', CHILD, 'child-1')) + h.send(turn('turn/started', CHILD, 'child-1')) + h.send(activity('parent-1', 'completed')) + expect(h.state(parent)).toBe('working') + expect(h.tracker.state?.tasks).toHaveLength(1) + if (parent === 'parent-2') { + expect(h.state('parent-1')).toBe('completed') + } + h.send(turn('turn/completed', CHILD, 'child-2')) + expect(h.state(parent)).toBe('completed') + expect(h.tracker.state).toBeNull() + h.dispose() + } + ) + + it('keeps an idle message from creating execution on either surface', () => { + const h = harness() + firstRun(h) + h.send(turn('turn/completed', PRIMARY, 'parent-1')) + h.send(activity('parent-2', 'interacted')) + h.send(turn('turn/completed', PRIMARY, 'parent-2')) + expect(h.state('parent-1')).toBe('completed') + expect(h.state('parent-2')).toBeUndefined() + expect(h.tracker.state).toBeNull() + h.dispose() + }) + + it('keeps a late completion activity from retargeting a queued follow-up', () => { + const h = harness() + firstRun(h) + h.send(turn('turn/completed', PRIMARY, 'parent-1')) + h.send(activity('parent-2', 'interacted')) + h.send(activity('parent-1', 'completed')) + h.send(turn('turn/started', CHILD, 'child-2')) + expect(h.state('parent-1')).toBe('completed') + expect(h.state('parent-2')).toBe('working') + expect(h.tracker.state?.tasks).toHaveLength(1) + h.dispose() + }) + + it('keeps a message to a running child in the original execution group', () => { + const h = harness() + h.send(turn('turn/started', PRIMARY, 'parent-1')) + h.send(turn('turn/started', CHILD, 'child-1')) + h.send(activity('parent-1')) + h.send(turn('turn/completed', PRIMARY, 'parent-1')) + h.send(turn('turn/started', PRIMARY, 'parent-2')) + h.send(activity('parent-2', 'interacted')) + h.send(turn('turn/started', CHILD, 'child-1')) + h.send(turn('turn/completed', PRIMARY, 'parent-2')) + expect(h.tracker.state?.tasks).toHaveLength(1) + expect(h.state('parent-2')).toBeUndefined() + h.send(turn('turn/completed', CHILD, 'child-1')) + expect(h.state('parent-1')).toBe('completed') + expect(h.state('parent-2')).toBeUndefined() + expect(h.tracker.state).toBeNull() + h.dispose() + }) + + it('does not expose pending child settlement before journal admission and retries the same fact', () => { + const h = harness() + h.send(turn('turn/started', PRIMARY, 'parent-1')) + h.send(activity('parent-1')) + h.send(turn('turn/started', CHILD, 'child-1')) + h.send(turn('turn/completed', PRIMARY, 'parent-1')) + const complete = turn('turn/completed', CHILD, 'child-1') + h.refuse(true) + expect(h.send(complete)).toEqual({ accepted: false, reason: 'backpressure' }) + expect(h.tracker.state?.tasks).toHaveLength(1) + expect(h.state('parent-1')).toBe('working') + h.refuse(false) + expect(h.send(complete)).toEqual({ accepted: true }) + expect(h.state('parent-1')).toBe('completed') + expect(h.tracker.state).toBeNull() + h.dispose() + }) +}) diff --git a/src/main/codex/codex-subagent-executions.test.ts b/src/main/codex/codex-subagent-executions.test.ts new file mode 100644 index 00000000000..aceff7ab465 --- /dev/null +++ b/src/main/codex/codex-subagent-executions.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it } from 'vitest' +import { CodexSubagentExecutions } from './codex-subagent-executions' + +describe('CodexSubagentExecutions retention and identity', () => { + it('bounds settled history through repeated execution without evicting live children', () => { + const executions = new CodexSubagentExecutions() + executions.register('long-lived', 'long-lived', 'parent') + executions.observeTurn('long-lived', 'long-lived-turn', 'working') + for (let index = 0; index < 1_000; index++) { + const id = `child-${index}` + executions.observeTurn(id, id, 'working') + executions.register(id, id, 'parent') + executions.observeTurn(id, id, 'completed') + } + expect(executions.workingChildren().map((child) => child.agentThreadId)).toEqual(['long-lived']) + expect(Reflect.get(executions, 'children').size).toBeLessThanOrEqual(128) + expect(Reflect.get(executions, 'settledTurns').size).toBeLessThanOrEqual(256) + }) + + it('retains early live owner events at capacity and makes room only after settlement', () => { + const executions = new CodexSubagentExecutions() + for (let index = 0; index < 128; index++) { + executions.observeTurn(`child-${index}`, `turn-${index}`, 'working') + } + expect(executions.observeTurn('overflow', 'overflow', 'working')).toBeNull() + for (let index = 0; index < 128; index++) { + executions.register(`child-${index}`, `child-${index}`, 'parent') + } + expect(executions.workingChildren()).toHaveLength(128) + executions.observeTurn('child-0', 'turn-0', 'completed') + expect(executions.observeTurn('overflow', 'overflow', 'working')).not.toBeNull() + executions.register('overflow', 'overflow', 'parent') + expect(executions.workingChildren()).toHaveLength(128) + }) + + it('corrects an unverifiable execution with its own terminal event and ignores stale starts', () => { + const executions = new CodexSubagentExecutions() + executions.register('child', 'child', 'parent') + executions.observeTurn('child', 'turn', 'working') + executions.settleSession() + expect(executions.workingChildren()).toEqual([]) + expect(executions.observeTurn('child', 'turn', 'working')).toBeNull() + expect(executions.observeTurn('child', 'turn', 'completed')?.execution.state).toBe('completed') + executions.observeTurn('child', 'new-turn', 'working') + executions.observeTurn('child', 'turn', 'failed') + expect(executions.workingChildren()[0]?.execution?.turnId).toBe('new-turn') + executions.clear() + expect(Reflect.get(executions, 'children').size).toBe(0) + expect(Reflect.get(executions, 'settledTurns').size).toBe(0) + }) +}) diff --git a/src/main/codex/codex-subagent-executions.ts b/src/main/codex/codex-subagent-executions.ts new file mode 100644 index 00000000000..cd33b4eb1d5 --- /dev/null +++ b/src/main/codex/codex-subagent-executions.ts @@ -0,0 +1,144 @@ +import type { NativeChatSubagentState } from '../../shared/native-chat-types' +import { MAX_SUBAGENT_FIELD_CHARS } from '../../shared/native-chat-subagent-summary' + +const MAX_CHILDREN = 128 +const MAX_SETTLED_TURNS = 256 + +export type CodexChildExecution = { + turnId: string + state: NativeChatSubagentState +} + +export type CodexExecutionChild = { + agentThreadId: string + registered: boolean + label: string | null + parentTurnId: string | null + execution: CodexChildExecution | null +} + +/** Child turn events own execution; activity items only identify the child. */ +export class CodexSubagentExecutions { + private readonly children = new Map() + private readonly settledTurns = new Map() + + register( + agentThreadId: string, + label: string | null, + parentTurnId: string | null | undefined + ): CodexExecutionChild | undefined { + const child = this.child(agentThreadId) + if (!child) { + return undefined + } + if (!child.registered || parentTurnId !== undefined) { + child.parentTurnId = parentTurnId ?? null + } + child.registered = true + // Retain one overflow unit so the journal can append its per-row truncation marker. + child.label ??= + label + ?.trim() + .replace(/\s+/g, ' ') + .slice(0, MAX_SUBAGENT_FIELD_CHARS + 1) || null + return child + } + + observeTurn( + agentThreadId: string, + turnId: string, + state: NativeChatSubagentState + ): { child: CodexExecutionChild; execution: CodexChildExecution } | null { + const key = JSON.stringify([agentThreadId, turnId]) + const settled = this.settledTurns.get(key) + if (state === 'working' && settled !== undefined) { + return null + } + const child = this.child(agentThreadId) + if (!child) { + return null + } + if ( + state === 'working' && + child.execution?.turnId === turnId && + child.execution.state !== 'working' + ) { + return null + } + const execution = { turnId, state: settled ?? state } + if (state !== 'working') { + this.settledTurns.set(key, execution.state) + while (this.settledTurns.size > MAX_SETTLED_TURNS) { + const oldest = this.settledTurns.keys().next().value + if (oldest === undefined) { + break + } + this.settledTurns.delete(oldest) + } + } + if (state === 'working' || !child.execution || child.execution.turnId === turnId) { + child.execution = execution + } + return { child, execution } + } + + /** Survives the child's turn, so a row outliving that turn can still name it. */ + label(agentThreadId: string): string | null { + return this.children.get(agentThreadId)?.label ?? null + } + + workingChildren(): CodexExecutionChild[] { + return [...this.children.values()].filter( + (child) => child.registered && child.execution?.state === 'working' + ) + } + + settleSession(): void { + for (const child of this.children.values()) { + if (child.execution?.state === 'working') { + child.execution = { ...child.execution, state: 'unverifiable' } + } + } + } + + clear(): void { + this.children.clear() + this.settledTurns.clear() + } + + private child(agentThreadId: string): CodexExecutionChild | undefined { + const existing = this.children.get(agentThreadId) + if (existing) { + return existing + } + if (this.children.size >= MAX_CHILDREN) { + const settled = [...this.children].find(([, child]) => child.execution?.state !== 'working') + if (!settled) { + return undefined + } + this.children.delete(settled[0]) + } + const child: CodexExecutionChild = { + agentThreadId, + registered: false, + label: null, + parentTurnId: null, + execution: null + } + this.children.set(agentThreadId, child) + return child + } +} + +export function codexChildTurnState(status: unknown): NativeChatSubagentState { + if (status === 'completed') { + return 'completed' + } + if (status === 'interrupted') { + return 'stopped' + } + if (status === 'failed') { + return 'failed' + } + return 'unverifiable' +} diff --git a/src/main/codex/codex-subagent-group-body.ts b/src/main/codex/codex-subagent-group-body.ts new file mode 100644 index 00000000000..c2157b0e7eb --- /dev/null +++ b/src/main/codex/codex-subagent-group-body.ts @@ -0,0 +1,51 @@ +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import type { NativeChatSubagentEntry } from '../../shared/native-chat-types' +import { + MAX_SUBAGENT_FIELD_CHARS, + subagentGroupFallbackText +} from '../../shared/native-chat-subagent-summary' + +/** The roster row: the structured block plus the plain sentence an older client + * renders in its place. A message whose only block is the new variant would + * reach such a client with nothing it can draw. */ +export function codexSubagentGroupBody( + groupId: string, + agents: readonly NativeChatSubagentEntry[] +): AgentJournalItemBody { + const bounded = agents.map((agent, index) => ({ + ...agent, + id: boundSubagentField(agent.id, index), + label: boundSubagentField(agent.label, index) + })) + return { + kind: 'message', + role: 'system', + blocks: [ + { type: 'text', text: subagentGroupFallbackText(bounded) }, + { type: 'subagent-group', groupId, agents: bounded } + ] + } +} + +/** `id` and `label` are provider strings, so they take the bound both readers of + * this row already clip them to. A plain length check, not the tool-output + * bound: that one digests the whole value before it checks the length, and this + * runs twice per child on every streamed token-usage frame. + * + * A clip is not identity-preserving, so a clipped value carries the child's + * index: two ids sharing a long prefix collapse to one React key, and + * `claimLabel` writes its ordinal at the very tail the clip removes. The index + * is reserved out of the bound, not appended to it, because both readers + * re-clip to the same cap and would cut a suffix that overflowed it. */ +export function boundSubagentField(value: string, index: number): string { + if (value.length <= MAX_SUBAGENT_FIELD_CHARS) { + return value + } + const suffix = `…~${index}` + const keep = MAX_SUBAGENT_FIELD_CHARS - suffix.length + // Slicing UTF-16 units can split a surrogate pair; a lone surrogate is + // malformed in a durable row and lossy through any non-JSON UTF-8 hop. + const last = value.charCodeAt(keep - 1) + const end = last >= 0xd800 && last <= 0xdbff ? keep - 1 : keep + return `${value.slice(0, end)}${suffix}` +} diff --git a/src/main/codex/codex-subagent-roster.test.ts b/src/main/codex/codex-subagent-roster.test.ts index 2f20c9df9bf..9d3e2c690c4 100644 --- a/src/main/codex/codex-subagent-roster.test.ts +++ b/src/main/codex/codex-subagent-roster.test.ts @@ -83,6 +83,22 @@ function deliver( item: CodexThreadItem, turnId: string | null = TURN ): void { + // The fixture includes the child's owner event separately from its activity metadata. + const state = + item.kind === 'started' + ? 'working' + : item.kind === 'completed' + ? 'completed' + : item.kind === 'interrupted' + ? 'stopped' + : null + if (state && typeof item.agentThreadId === 'string') { + roster.handleTurn({ + threadId: item.agentThreadId, + turnId: `execution:${item.agentThreadId}`, + state + }) + } // Every activity item reaches the wire twice: item/started, then item/completed. roster.handleItem({ threadId: THREAD, turnId, item }) roster.handleItem({ threadId: THREAD, turnId, item }) @@ -270,7 +286,7 @@ describe('CodexSubagentRoster', () => { expect(appended).toHaveLength(1) }) - it('rule 2 — a first event of any kind creates the entry in the state it implies', () => { + it('registers a child after its owner already reported completion', () => { const { roster, agents } = createHarness() deliver( @@ -374,7 +390,7 @@ describe('CodexSubagentRoster', () => { deliver( roster, - activity({ kind: 'interacted', agentThreadId: 'child-1', agentPath: '/root/read' }) + activity({ kind: 'started', agentThreadId: 'child-1', agentPath: '/root/read' }) ) roster.settleSession() const afterFirstSweep = appended.length @@ -676,6 +692,7 @@ describe('CodexSubagentRoster', () => { now: () => 1_000 }) const item = activity({ kind: 'started', agentThreadId: 'child-1', agentPath: '/root/read' }) + roster.handleTurn({ threadId: 'child-1', turnId: 'child-turn', state: 'working' }) expect(roster.handleItem({ threadId: THREAD, turnId: TURN, item })).toEqual(refusal) @@ -724,6 +741,7 @@ describe('CodexSubagentRoster', () => { activeTurn: () => TURN, now: () => 1_000 }) + roster.handleTurn({ threadId: 'child-1', turnId: 'child-turn', state: 'working' }) roster.handleItem({ threadId: THREAD, turnId: TURN, @@ -758,6 +776,7 @@ describe('CodexSubagentRoster', () => { activeTurn: () => TURN }) + roster.handleTurn({ threadId: 'child-1', turnId: 'child-turn', state: 'working' }) expect( roster.handleItem({ threadId: THREAD, diff --git a/src/main/codex/codex-subagent-roster.ts b/src/main/codex/codex-subagent-roster.ts index 257fe705764..a2b58e1a6ce 100644 --- a/src/main/codex/codex-subagent-roster.ts +++ b/src/main/codex/codex-subagent-roster.ts @@ -1,11 +1,6 @@ // The Codex subagent roster: one journal row per spawn group, revised in place. // -// There is no snapshot to read. `agentsStates` arrived empty in the live probe -// and children get no `thread/started`, so the roster is -// accumulated purely from `subAgentActivity` items — each of which arrives TWICE -// (`item/started` and `item/completed`). Every transition here is therefore -// idempotent, and a terminal state latches: duplicate and out-of-order delivery -// must not resurrect a settled child. +// Activity supplies membership; child turn events supply execution state. // // KNOWN LIMITATION: `groups` is process-local and is never seeded from the // journal, while the row's identity is keyed on the group id alone. So once a @@ -18,28 +13,32 @@ // thread, and a real turn id is assumed freshly minted per turn. Seeding from // the journal is the fix. +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { isTerminalSubagentState } from '../../shared/native-chat-subagent-summary' import type { - AgentJournalItemBody, - AgentJournalItemIdentity -} from '../../shared/agent-session-journal-types' -import { - canReplaceSubagentState, - isTerminalSubagentState, - MAX_SUBAGENT_FIELD_CHARS, - subagentGroupFallbackText -} from '../../shared/native-chat-subagent-summary' -import type { NativeChatSubagentEntry } from '../../shared/native-chat-types' + NativeChatSubagentEntry, + NativeChatSubagentState +} from '../../shared/native-chat-types' import type { StructuredAgentSessionEventSink, StructuredAgentSessionSinkAdmission } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' import { codexSubagentLabel, - codexSubagentStateForKind, isCodexRootAgentActivity, readCodexSubagentActivity, readCodexThreadTokenTotal } from './codex-subagent-activity' +import { + CodexSubagentExecutions, + codexChildTurnState, + type CodexChildExecution, + type CodexExecutionChild +} from './codex-subagent-executions' +import { readRecord } from './codex-item-field-readers' +import { readCodexTurnId } from './codex-structured-thread-facts' +import { codexSubagentGroupBody } from './codex-subagent-group-body' +export { codexSubagentGroupBody } from './codex-subagent-group-body' import type { CodexThreadItem } from './codex-structured-item-translation' import { MAX_CODEX_SUBAGENT_GROUPS, @@ -51,15 +50,12 @@ const ADMITTED: StructuredAgentSessionSinkAdmission = { accepted: true } /** The turn a group belongs to when Codex reports activity outside any turn. * Mirrors the generic-frame bucket name so the two read alike in the journal. */ -const OUTSIDE_TURN = 'outside-turn' - -const UNLABELLED_AGENT = 'subagent' - type RosterGroup = { groupId: string identity: AgentJournalItemIdentity /** Insertion order is the display order; the map holds the state. */ entries: Map + executionTurns: Map /** Times each label has been claimed, so a repeat gets an ordinal suffix. */ labelCounts: Map /** Last body written, so an idempotent replay writes no new revision. */ @@ -70,7 +66,7 @@ type RosterGroup = { * tree rooted at the parent thread, so every child of one turn shares a row * no matter which thread's stream carried its activity item. */ export function codexSubagentGroupId(threadId: string, turnId: string | null): string { - return `${threadId}:${turnId ?? OUTSIDE_TURN}` + return `${threadId}:${turnId ?? 'outside-turn'}` } /** Durable journal identity for the group's row — stable across revisions and @@ -85,6 +81,7 @@ export type CodexSubagentRosterDeps = { primaryThreadId: () => string | null activeTurn: (threadId: string) => string | null now?: () => number + executions?: CodexSubagentExecutions } export class CodexSubagentRoster { @@ -95,9 +92,11 @@ export class CodexSubagentRoster { * the map itself is LRU-capped in `handleTokenUsage`. */ private readonly tokensByThread = new Map() private readonly now: () => number + private readonly executions: CodexSubagentExecutions constructor(private readonly deps: CodexSubagentRosterDeps) { this.now = deps.now ?? (() => Date.now()) + this.executions = deps.executions ?? new CodexSubagentExecutions() } /** Consume a `subAgentActivity` item. Returns null when the item is not one. */ @@ -111,42 +110,81 @@ export class CodexSubagentRoster { return null } // The root node is the parent turn itself, not a child it spawned. - if (isCodexRootAgentActivity(activity)) { + if ( + activity.agentThreadId === this.deps.primaryThreadId() || + isCodexRootAgentActivity(activity) + ) { return ADMITTED } - const group = this.groupFor(input.threadId, input.turnId) - const existing = group.entries.get(activity.agentThreadId) - const state = codexSubagentStateForKind(activity.kind) - if (!existing) { - // Rule: the first event for a child may be ANY kind. An `interacted` or - // `completed` with no prior `started` creates the entry in the state its - // kind implies rather than being dropped for lacking a roster row. - if (group.entries.size >= MAX_CODEX_SUBAGENTS_PER_GROUP) { - return ADMITTED - } - const now = this.now() - group.entries.set(activity.agentThreadId, { - id: activity.agentThreadId, - label: this.claimLabel(group, codexSubagentLabel(activity)), - state, - startedAt: now, - ...(isTerminalSubagentState(state) ? { settledAt: now } : {}) - }) - } else if (canReplaceSubagentState(existing.state, state)) { - // A child's own verdict latches. Re-applying the same non-terminal state - // is a no-op, which is what makes the duplicate `item/started` + - // `item/completed` delivery idempotent. `unverifiable` does not latch: a - // child swept when contact was lost can still report what it actually did - // if contact returns. - group.entries.set(activity.agentThreadId, { - ...existing, - state, - ...(isTerminalSubagentState(state) ? { settledAt: this.now() } : {}) - }) + const child = this.executions.register( + activity.agentThreadId, + codexSubagentLabel(activity), + activity.kind === 'started' || activity.kind === 'interacted' ? input.turnId : undefined + ) + if (!child?.execution) { + return ADMITTED + } + const group = + this.executionGroup(child.agentThreadId, child.execution.turnId) ?? + this.groupFor(input.threadId, input.turnId) + if (!group.entries.has(child.agentThreadId)) { + this.recordExecution(group, child, child.execution) } return this.write(group) } + handleTurnEvent(event: { + method: string + threadId: string + params: unknown + }): StructuredAgentSessionSinkAdmission { + const turnId = readCodexTurnId(event.params) + return turnId + ? this.handleTurn({ + threadId: event.threadId, + turnId, + state: + event.method === 'turn/started' + ? 'working' + : codexChildTurnState(readRecord(readRecord(event.params).turn).status) + }) + : ADMITTED + } + + handleTurn(input: { + threadId: string + turnId: string + state: NativeChatSubagentState + }): StructuredAgentSessionSinkAdmission { + if (input.threadId === this.deps.primaryThreadId()) { + return ADMITTED + } + const observed = this.executions.observeTurn(input.threadId, input.turnId, input.state) + if (!observed || !observed.child.registered) { + return ADMITTED + } + const { child, execution } = observed + if (input.state === 'working') { + const parent = this.deps.primaryThreadId() ?? input.threadId + const group = + this.executionGroup(child.agentThreadId, execution.turnId) ?? + this.groupFor(parent, this.deps.activeTurn(parent) ?? child.parentTurnId) + this.recordExecution(group, child, execution) + return this.write(group) + } + for (const group of this.groups.values()) { + if (group.executionTurns.get(input.threadId) !== input.turnId) { + continue + } + this.recordExecution(group, child, execution) + const admission = this.write(group) + if (!admission.accepted) { + return admission + } + } + return ADMITTED + } + /** Consume `thread/tokenUsage/updated`. Returns null when the params are not one. */ handleTokenUsage(params: unknown): StructuredAgentSessionSinkAdmission | null { const usage = readCodexThreadTokenTotal(params) @@ -187,6 +225,7 @@ export class CodexSubagentRoster { * routinely outlive their turn and keep reporting into the same group. */ settleSession(): StructuredAgentSessionSinkAdmission { + this.executions.settleSession() for (const group of this.groups.values()) { const admission = this.sweep(group) if (!admission.accepted) { @@ -233,6 +272,7 @@ export class CodexSubagentRoster { groupId, identity: codexSubagentGroupIdentity(groupId), entries: new Map(), + executionTurns: new Map(), labelCounts: new Map(), lastSerialized: null } @@ -247,15 +287,46 @@ export class CodexSubagentRoster { return group } + private executionGroup(threadId: string, turnId: string): RosterGroup | undefined { + return [...this.groups.values()].find((group) => group.executionTurns.get(threadId) === turnId) + } + /** Two children can share a trailing path segment; the ordinal keeps their * rows apart without inventing a name the provider never sent. */ private claimLabel(group: RosterGroup, label: string | null): string { - const base = label ?? UNLABELLED_AGENT + const base = label ?? 'subagent' const seen = group.labelCounts.get(base) ?? 0 group.labelCounts.set(base, seen + 1) return seen === 0 ? base : `${base} ${seen + 1}` } + private recordExecution( + group: RosterGroup, + child: CodexExecutionChild, + execution: CodexChildExecution | null + ): void { + const existing = group.entries.get(child.agentThreadId) + if (!existing && group.entries.size >= MAX_CODEX_SUBAGENTS_PER_GROUP) { + return + } + const turnId = execution?.turnId ?? null + const state = execution?.state ?? 'unverifiable' + const sameTurn = existing && group.executionTurns.get(child.agentThreadId) === turnId + if (sameTurn && existing.state === state) { + return + } + const now = this.now() + group.executionTurns.set(child.agentThreadId, turnId) + group.entries.set(child.agentThreadId, { + id: child.agentThreadId, + label: existing?.label ?? this.claimLabel(group, child.label), + state, + startedAt: sameTurn ? existing.startedAt : now, + ...(isTerminalSubagentState(state) ? { settledAt: now } : {}), + ...(existing?.tokens !== undefined ? { tokens: existing.tokens } : {}) + }) + } + private write(group: RosterGroup): StructuredAgentSessionSinkAdmission { const agents = [...group.entries].map(([id, entry]) => { const tokens = this.tokensByThread.get(id) @@ -300,48 +371,3 @@ export class CodexSubagentRoster { return published } } - -/** The roster row: the structured block plus the plain sentence an older client - * renders in its place. A message whose only block is the new variant would - * reach such a client with nothing it can draw. */ -export function codexSubagentGroupBody( - groupId: string, - agents: readonly NativeChatSubagentEntry[] -): AgentJournalItemBody { - const bounded = agents.map((agent, index) => ({ - ...agent, - id: boundSubagentField(agent.id, index), - label: boundSubagentField(agent.label, index) - })) - return { - kind: 'message', - role: 'system', - blocks: [ - { type: 'text', text: subagentGroupFallbackText(bounded) }, - { type: 'subagent-group', groupId, agents: bounded } - ] - } -} - -/** `id` and `label` are provider strings, so they take the bound both readers of - * this row already clip them to. A plain length check, not the tool-output - * bound: that one digests the whole value before it checks the length, and this - * runs twice per child on every streamed token-usage frame. - * - * A clip is not identity-preserving, so a clipped value carries the child's - * index: two ids sharing a long prefix collapse to one React key, and - * `claimLabel` writes its ordinal at the very tail the clip removes. The index - * is reserved out of the bound, not appended to it, because both readers - * re-clip to the same cap and would cut a suffix that overflowed it. */ -function boundSubagentField(value: string, index: number): string { - if (value.length <= MAX_SUBAGENT_FIELD_CHARS) { - return value - } - const suffix = `…~${index}` - const keep = MAX_SUBAGENT_FIELD_CHARS - suffix.length - // Slicing UTF-16 units can split a surrogate pair; a lone surrogate is - // malformed in a durable row and lossy through any non-JSON UTF-8 hop. - const last = value.charCodeAt(keep - 1) - const end = last >= 0xd800 && last <= 0xdbff ? keep - 1 : keep - return `${value.slice(0, end)}${suffix}` -} diff --git a/src/main/codex/codex-turn-ordinals.test.ts b/src/main/codex/codex-turn-ordinals.test.ts new file mode 100644 index 00000000000..5ae12011a85 --- /dev/null +++ b/src/main/codex/codex-turn-ordinals.test.ts @@ -0,0 +1,59 @@ +import { expect, it } from 'vitest' +import { + CodexTurnOrdinals, + MAX_CODEX_TURN_ORDINAL_BYTES, + MAX_CODEX_TURN_ORDINAL_ENTRIES +} from './codex-turn-ordinals' + +it('does not rescan the forgotten window for each new streamed item', () => { + const ordinals = new CodexTurnOrdinals() + for (let index = 0; index < MAX_CODEX_TURN_ORDINAL_ENTRIES; index += 1) { + ordinals.ordinalFor('thread', String(index), 'item') + ordinals.forgetTurn('thread', String(index)) + } + const turns = (ordinals as unknown as { turns: Map }).turns + let reads = 0 + for (const turn of turns.values()) { + let active = turn.active + Object.defineProperty(turn, 'active', { + get() { + reads += 1 + return active + }, + set(value: boolean) { + active = value + } + }) + } + for (let index = 0; index < 1000; index += 1) { + expect(ordinals.ordinalFor('thread', 'live', String(index))).toBe(index) + } + expect(reads).toBe(0) + expect(ordinals.forgottenTurnCount).toBe(MAX_CODEX_TURN_ORDINAL_ENTRIES) +}) + +it('retains ordinal continuity on late reactivation and evicts oldest forgotten turns', () => { + const ordinals = new CodexTurnOrdinals() + expect(ordinals.ordinalFor('t', 'first', 'a')).toBe(0) + ordinals.forgetTurn('t', 'first') + expect(ordinals.ordinalFor('t', 'first', 'b')).toBe(1) + expect(ordinals.forgottenTurnCount).toBe(0) + ordinals.forgetTurn('t', 'first') + for (let index = 0; index < MAX_CODEX_TURN_ORDINAL_ENTRIES; index += 1) { + ordinals.ordinalFor('t', String(index), 'a') + ordinals.forgetTurn('t', String(index)) + } + expect(ordinals.forgottenTurnCount).toBe(MAX_CODEX_TURN_ORDINAL_ENTRIES) + expect(ordinals.ordinalFor('t', 'first', 'c')).toBe(0) +}) + +it('keeps byte eviction bounded for active and forgotten turns', () => { + const ordinals = new CodexTurnOrdinals() + for (let index = 0; index < 4000; index += 1) { + ordinals.ordinalFor('thread', 'live', `${index}-${'x'.repeat(240)}`) + } + expect(ordinals.bytes).toBeLessThanOrEqual(MAX_CODEX_TURN_ORDINAL_BYTES) + ordinals.forgetTurn('thread', 'live') + expect(ordinals.bytes).toBeLessThan(100) + expect(ordinals.forgottenTurnCount).toBe(1) +}) diff --git a/src/main/codex/codex-turn-ordinals.ts b/src/main/codex/codex-turn-ordinals.ts index e2b4132d2b2..89ed72666db 100644 --- a/src/main/codex/codex-turn-ordinals.ts +++ b/src/main/codex/codex-turn-ordinals.ts @@ -14,15 +14,10 @@ export class CodexTurnOrdinals { { assigned: Map; next: number; active: boolean } >() private retainedBytes = 0 + private readonly forgottenTurns = new Set() get forgottenTurnCount(): number { - let count = 0 - for (const turn of this.turns.values()) { - if (!turn.active) { - count += 1 - } - } - return count + return this.forgottenTurns.size } get bytes(): number { @@ -48,12 +43,13 @@ export class CodexTurnOrdinals { private trimForgotten(): void { while (this.forgottenTurnCount > MAX_CODEX_TURN_ORDINAL_ENTRIES) { - const oldest = [...this.turns.entries()].find(([, turn]) => !turn.active)?.[0] + const oldest = this.forgottenTurns.values().next().value if (!oldest) { break } const removed = this.turns.get(oldest) this.turns.delete(oldest) + this.forgottenTurns.delete(oldest) if (removed) { this.retainedBytes = Math.max( 0, @@ -68,7 +64,7 @@ export class CodexTurnOrdinals { private trimBytes(currentTurnKey: string): void { this.trimForgotten() while (this.retainedBytes > MAX_CODEX_TURN_ORDINAL_BYTES) { - const forgotten = [...this.turns.entries()].find(([, turn]) => !turn.active)?.[0] + const forgotten = this.forgottenTurns.values().next().value const oldest = forgotten ?? this.turns.keys().next().value if (typeof oldest !== 'string') { break @@ -90,6 +86,7 @@ export class CodexTurnOrdinals { break } this.turns.delete(oldest) + this.forgottenTurns.delete(oldest) this.retainedBytes = Math.max( 0, this.retainedBytes - @@ -112,6 +109,7 @@ export class CodexTurnOrdinals { this.turns.set(turnKey, turn) } turn.active = true + this.forgottenTurns.delete(turnKey) } const itemKey = this.keyPart(codexItemId) const existing = turn.assigned.get(itemKey) @@ -136,6 +134,8 @@ export class CodexTurnOrdinals { ) turn.assigned = new Map() turn.active = false + this.forgottenTurns.delete(turnKey) + this.forgottenTurns.add(turnKey) this.retainedBytes = Math.max(0, this.retainedBytes - assignedBytes) this.turns.delete(turnKey) this.turns.set(turnKey, turn) diff --git a/src/main/codex/config-toml-hook-trust-edit.ts b/src/main/codex/config-toml-hook-trust-edit.ts index 9d6e5dbc2d5..a8487012f05 100644 --- a/src/main/codex/config-toml-hook-trust-edit.ts +++ b/src/main/codex/config-toml-hook-trust-edit.ts @@ -56,7 +56,10 @@ function upsertTrustBlocks( hash: string, explicitEnabled?: boolean ): string { - const ranges = getUniqueTrustBlockRanges(content, keys) + const ranges = findHookTrustBlockRanges( + content, + new Set(keys.map(normalizeCodexHookTrustLookupKey)) + ) if (ranges.length === 0) { return appendTrustBlocks(content, keys, hash, explicitEnabled ?? true) } @@ -74,21 +77,6 @@ function upsertTrustBlocks( return deduped + content.slice(cursor) } -function getUniqueTrustBlockRanges( - content: string, - keys: readonly string[] -): HookTrustBlockRange[] { - const normalizedKeys = new Set(keys.map(normalizeCodexHookTrustLookupKey)) - return findHookTrustBlockRanges(content, normalizedKeys) - .filter( - (range, index, ranges) => - ranges.findIndex( - (candidate) => candidate.start === range.start && candidate.end === range.end - ) === index - ) - .sort((left, right) => left.start - right.start) -} - function isBlockDisabled(content: string, range: HookTrustBlockRange): boolean { const block = content.slice(range.headerLineEnd, range.end) const enabledMatch = /^[ \t]*enabled[ \t]*=[ \t]*(true|false)[ \t\r]*(?:#.*)?$/m.exec(block) diff --git a/src/main/codex/config-toml-hook-trust-scaling.test.ts b/src/main/codex/config-toml-hook-trust-scaling.test.ts new file mode 100644 index 00000000000..bb744fa4b16 --- /dev/null +++ b/src/main/codex/config-toml-hook-trust-scaling.test.ts @@ -0,0 +1,72 @@ +import type * as HookTrustBlocks from './config-toml-hook-trust-blocks' +import { expect, it, vi } from 'vitest' +import { upsertHookTrustContent } from './config-toml-hook-trust-edit' + +const counts = vi.hoisted(() => ({ starts: 0 })) +vi.mock('./config-toml-hook-trust-blocks', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + findHookTrustBlockRanges: (...args: Parameters) => + actual.findHookTrustBlockRanges(...args).map((range) => ({ + ...range, + get start() { + counts.starts += 1 + return range.start + } + })) + } +}) + +it('consumes monotonically scanned trust ranges without pairwise deduplication', () => { + const header = '[hooks.state."/foo/hooks.json:pre_tool_use:0:0"]' + const content = `${header}\nenabled = true\ntrusted_hash = "old"\n`.repeat(1000) + counts.starts = 0 + const result = upsertHookTrustContent(content, [ + { + sourcePath: '/foo/hooks.json', + eventLabel: 'pre_tool_use', + groupIndex: 0, + handlerIndex: 0, + command: '/bin/echo hi', + trustedHash: 'updated' + } + ]) + expect(counts.starts).toBeLessThan(5000) + expect(result.split(header)).toHaveLength(2) + expect(result).toContain('trusted_hash = "updated"') +}) + +// Dropping the old dedup+sort is only sound because the scanner advances its +// cursor past each block it emits. Pin that precondition: if a future scanner +// change lets ranges repeat or overlap, the upsert below would delete or widen +// a neighbouring trust block instead of rewriting just the matched one. +it('emits trust ranges with strictly ascending, non-overlapping spans', async () => { + const { findHookTrustBlockRanges } = await vi.importActual( + './config-toml-hook-trust-blocks' + ) + const key = '/foo/hooks.json:pre_tool_use:0:0' + const header = `[hooks.state."${key}"]` + const contents = [ + '', + header, + `${header}\n${header}\n`, + `${header}\nenabled = true\n`.repeat(50), + `${header}\r\nenabled = true\r\n`.repeat(3), + `[x]\nv = """\n${header}\n"""\n${header}\nenabled = true\n`, + `[x]\na = [\n${header}\n]\n${header}\nenabled = true\n`, + `${header}\nenabled = true\n[[arr]]\nz = 1\n${header}\n` + ] + for (const content of contents) { + const ranges = findHookTrustBlockRanges(content, new Set([key])) + for (const [index, range] of ranges.entries()) { + expect(range.end).toBeGreaterThanOrEqual(range.start) + expect(range.end).toBeLessThanOrEqual(content.length) + if (index > 0) { + expect(ranges[index - 1].start).toBeLessThan(range.start) + expect(ranges[index - 1].end).toBeLessThanOrEqual(range.start) + } + } + expect(new Set(ranges.map((range) => `${range.start}:${range.end}`)).size).toBe(ranges.length) + } +}) diff --git a/src/main/copilot/copilot-managed-script.ts b/src/main/copilot/copilot-managed-script.ts index 492caafc045..018b026c992 100644 --- a/src/main/copilot/copilot-managed-script.ts +++ b/src/main/copilot/copilot-managed-script.ts @@ -1,7 +1,8 @@ import { getSharedManagedScriptPath } from '../agent-hooks/installer-utils' import { buildPosixHookPayloadCapture, - buildPosixHookSpoolLines + buildPosixHookSpoolLines, + WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD } from '../agent-hooks/hook-stdin-contract' export function getManagedScriptFileName(): string { @@ -30,7 +31,7 @@ export function getManagedScript(target: 'local' | 'posix' = 'local'): string { // Why (#11549 class): missing Orca context means a user-wide hook fired outside an // Orca pane. ReadToEnd blocks forever if that caller abandons the pipe, so the guard // must run before the hook owns stdin; the payload would be discarded anyway. - 'if (-not $env:ORCA_AGENT_HOOK_PORT -or -not $env:ORCA_AGENT_HOOK_TOKEN -or -not $env:ORCA_PANE_KEY) { exit 0 }', + WINDOWS_POWERSHELL_HOOK_ENVIRONMENT_GUARD, '$inputData = [Console]::In.ReadToEnd()', 'if ([string]::IsNullOrWhiteSpace($inputData)) { exit 0 }', 'try {', diff --git a/src/main/daemon/daemon-request-router.ts b/src/main/daemon/daemon-request-router.ts index bb7d0d1a256..ae007366dd0 100644 --- a/src/main/daemon/daemon-request-router.ts +++ b/src/main/daemon/daemon-request-router.ts @@ -203,7 +203,11 @@ export class DaemonRequestRouter { await this.options.host.kill(sessionId, { immediate }) } catch (error) { if (!(canceledPendingSpawn && error instanceof SessionNotFoundError)) { - this.options.log.log('session-kill-failed', attribution) + this.options.log.log('session-kill-failed', { + ...attribution, + errorName: error instanceof Error ? error.name : typeof error, + error: error instanceof Error ? error.message : String(error) + }) throw error } } diff --git a/src/main/daemon/daemon-server-kill-attribution.test.ts b/src/main/daemon/daemon-server-kill-attribution.test.ts index 8fa6805e862..b2f506f42ed 100644 --- a/src/main/daemon/daemon-server-kill-attribution.test.ts +++ b/src/main/daemon/daemon-server-kill-attribution.test.ts @@ -85,7 +85,9 @@ describe('daemon kill attribution', () => { expect(killLog.log).toHaveBeenCalledWith('session-kill-failed', { sessionId: 'agent-session', immediate: true, - clientId: 'control-42' + clientId: 'control-42', + errorName: 'Error', + error: 'kill refused' }) expect(killLog.log).not.toHaveBeenCalledWith('session-killed', expect.anything()) }) diff --git a/src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts b/src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts new file mode 100644 index 00000000000..b7ce586668b --- /dev/null +++ b/src/main/daemon/pty-subprocess-io-failure-cleanup.test.ts @@ -0,0 +1,212 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as pty from 'node-pty' +import { createDaemonPtySubprocessHandle } from './pty-subprocess/subprocess-handle' +import { mockPtyProcess } from './pty-subprocess-test-harness' +import { TerminalHost } from './terminal-host' +import { HeadlessEmulator } from './headless-emulator' +import * as ptyJob from '../windows/windows-pty-job' + +vi.mock('./pty-subprocess/foreground-process-tracker', () => ({ + createPtyForegroundProcessTracker: () => ({ + recordOutput: vi.fn(), + markDead: vi.fn(), + getForegroundProcess: () => null + }) +})) +vi.mock('../pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: (_pid: number, fallback: () => void) => fallback() +})) +vi.mock('../pty-descendant-termination', () => ({ + killWithDescendantSweep: async (_pid: number, killRoot: () => void) => killRoot() +})) + +function createFixture() { + const proc = { + ...mockPtyProcess(4242), + destroy: vi.fn(), + pause: vi.fn(), + resume: vi.fn(), + clear: vi.fn() + } + const handle = createDaemonPtySubprocessHandle({ + process: proc as unknown as pty.IPty, + shellPath: 'bash', + spawnCwd: process.cwd(), + env: {}, + startupCommandDeliveredInShellArgs: false, + reportsChildExitStatus: true, + sessionId: 'io-failure', + startupAgentRecognition: null + }) + return { proc, handle } +} + +function failIo(fixture: ReturnType, operation: 'write' | 'resize') { + fixture.proc[operation].mockImplementation(() => { + throw new Error('transient native I/O failure') + }) + if (operation === 'write') { + fixture.handle.write('input') + } else { + fixture.handle.resize(100, 30) + } +} + +afterEach(() => vi.restoreAllMocks()) + +describe.each(['darwin', 'linux', 'win32'] as const)('%s native-handle contract', (platform) => { + const platformDescriptor = Object.getOwnPropertyDescriptor(process, 'platform')! + beforeEach(() => Object.defineProperty(process, 'platform', { value: platform })) + afterEach(() => Object.defineProperty(process, 'platform', platformDescriptor)) + + describe.each(['write', 'resize'] as const)('%s failure cleanup', (operation) => { + it('suppresses repeated native I/O failures while still delivering output and exit', () => { + const fixture = createFixture() + const onData = vi.fn() + const onExit = vi.fn() + fixture.handle.onData(onData) + fixture.handle.onExit(onExit) + failIo(fixture, operation) + fixture.handle.pause?.() + fixture.handle.resume?.() + expect(fixture.proc.pause).toHaveBeenCalledOnce() + expect(fixture.proc.resume).toHaveBeenCalledOnce() + fixture.handle.write('more input') + fixture.handle.resize(120, 40) + fixture.handle.clear?.() + expect(fixture.proc[operation]).toHaveBeenCalledOnce() + for (const suppressed of ['write', 'resize', 'clear'] as const) { + if (suppressed !== operation) { + expect(fixture.proc[suppressed]).not.toHaveBeenCalled() + } + } + fixture.proc._simulateData('still running') + expect(onData).toHaveBeenCalledWith('still running') + expect(onExit).not.toHaveBeenCalled() + fixture.proc._simulateExit(7) + expect(onExit).toHaveBeenCalledOnce() + fixture.handle.dispose() + }) + + it('blocks reentrant termination from an exit listener after I/O failure', () => { + const fixture = createFixture() + const signal = vi.spyOn(process, 'kill').mockReturnValue(true) + const nativeKill = fixture.proc.kill + failIo(fixture, operation) + fixture.handle.onExit(() => { + fixture.handle.kill() + fixture.handle.forceKill() + fixture.handle.signal('SIGTERM') + }) + fixture.proc._simulateExit(0) + expect(nativeKill).not.toHaveBeenCalled() + expect(signal).not.toHaveBeenCalled() + fixture.handle.dispose() + }) + + it.skipIf(platform !== 'win32')( + 'preserves ConPTY single-close ownership after I/O failure', + () => { + const fixture = createFixture() + const terminateJob = vi.spyOn(ptyJob, 'terminatePtyJob').mockReturnValue('terminated') + const signal = vi.spyOn(process, 'kill').mockReturnValue(true) + failIo(fixture, operation) + fixture.handle.kill() + fixture.handle.forceKill() + fixture.proc._simulateExit(137) + fixture.handle.dispose() + expect(fixture.proc.kill).toHaveBeenCalledOnce() + expect(terminateJob).toHaveBeenCalledOnce() + expect(signal).not.toHaveBeenCalled() + expect(fixture.proc.destroy).not.toHaveBeenCalled() + } + ) + + it('preserves early output and exit status while fencing listener cleanup', () => { + const fixture = createFixture() + const signal = vi.spyOn(process, 'kill').mockReturnValue(true) + failIo(fixture, operation) + fixture.proc._simulateData('final output') + fixture.proc._simulateExit(7) + const delivered: string[] = [] + fixture.handle.onData((data) => { + delivered.push(data) + fixture.handle.forceKill() + }) + fixture.handle.onExit((code) => delivered.push(`exit:${code}`)) + expect(delivered).toEqual(['final output', 'exit:7']) + expect(signal).not.toHaveBeenCalled() + fixture.handle.dispose() + }) + + it('keeps graceful and forced termination available until physical exit', () => { + const fixture = createFixture() + const originalKill = fixture.proc.kill + const signal = vi.spyOn(process, 'kill').mockReturnValue(true) + failIo(fixture, operation) + + fixture.handle.kill() + expect(originalKill).toHaveBeenCalledOnce() + // A fresh handle exercises force-kill without Windows double-close semantics. + const forced = createFixture() + failIo(forced, operation) + forced.handle.forceKill() + expect(signal).toHaveBeenCalledWith(4242, 'SIGKILL') + + forced.proc._simulateExit(137) + signal.mockClear() + forced.handle.forceKill() + forced.handle.signal('SIGTERM') + expect(signal).not.toHaveBeenCalled() + fixture.proc._simulateExit(0) + fixture.handle.dispose() + forced.handle.dispose() + }) + + it('reaps every session and native handle across 32 failed-I/O create/close cycles', async () => { + const emulatorDispose = vi.spyOn(HeadlessEmulator.prototype, 'dispose') + const signal = vi.spyOn(process, 'kill').mockReturnValue(true) + let fixture = createFixture() + const host = new TerminalHost({ spawnSubprocess: () => fixture.handle }) + try { + for (let index = 0; index < 32; index++) { + fixture = createFixture() + const sessionId = `io-failure-${index}` + const onExit = vi.fn() + await host.createOrAttach({ + sessionId, + cols: 80, + rows: 24, + streamClient: { onData: vi.fn(), onExit } + }) + failIo(fixture, operation) + signal.mockClear() + const closing = host.kill(sessionId, { immediate: true }) + // Capture rejection before assertions so a red run cannot leak an unhandled waiter. + const settled = closing.then( + () => null, + (error: unknown) => error ?? new Error('kill rejected') + ) + let killFailure: unknown = null + try { + expect(host.listSessions()).toHaveLength(1) + expect(fixture.proc.destroy).not.toHaveBeenCalled() + expect(onExit).not.toHaveBeenCalled() + await vi.waitFor(() => expect(signal).toHaveBeenCalledWith(4242, 'SIGKILL')) + } finally { + fixture.proc._simulateExit(137) + killFailure = await settled + } + expect(killFailure).toBeNull() + expect(host.listSessions()).toHaveLength(0) + expect(onExit).toHaveBeenCalledOnce() + expect(fixture.proc.destroy).toHaveBeenCalledOnce() + expect(emulatorDispose).toHaveBeenCalledTimes(index + 1) + } + } finally { + fixture.proc._simulateExit(137) + await host.dispose() + } + }) + }) +}) diff --git a/src/main/daemon/pty-subprocess-io-failure-native.test.ts b/src/main/daemon/pty-subprocess-io-failure-native.test.ts new file mode 100644 index 00000000000..b1835c068a7 --- /dev/null +++ b/src/main/daemon/pty-subprocess-io-failure-native.test.ts @@ -0,0 +1,97 @@ +import { fstatSync } from 'node:fs' +import * as pty from 'node-pty' +import { describe, expect, it, vi } from 'vitest' +import { createDaemonPtySubprocessHandle } from './pty-subprocess/subprocess-handle' +import { TerminalHost } from './terminal-host' + +const describePosix = process.platform === 'win32' ? describe.skip : describe + +describePosix('failed-I/O teardown with a real native PTY', () => { + it.each([ + ['write', false], + ['write', true], + ['resize', false], + ['resize', true] + ] as const)( + 'reaps real shells and master fds after %s failure (immediate=%s)', + async (operation, immediate) => { + for (let cycle = 0; cycle < 4; cycle++) { + const native = pty.spawn( + '/bin/sh', + [ + '-c', + 'printf "orca-cleanup-ready\\n"; while IFS= read -r line; do printf "reply:%s\\n" "$line"; done' + ], + { + cwd: process.cwd(), + cols: 80, + rows: 24, + env: { TERM: 'xterm-256color', PATH: '/usr/bin:/bin' } + } + ) + const fd = (native as pty.IPty & { fd: number }).fd + let exited = false + native.onExit(() => { + exited = true + }) + const handle = createDaemonPtySubprocessHandle({ + process: native, + shellPath: '/bin/sh', + spawnCwd: process.cwd(), + env: {}, + startupCommandDeliveredInShellArgs: false, + reportsChildExitStatus: true, + sessionId: 'native-io-failure', + startupAgentRecognition: null + }) + const host = new TerminalHost({ spawnSubprocess: () => handle }) + let output = '' + const onExit = vi.fn() + try { + await host.createOrAttach({ + sessionId: 'native-io-failure', + cols: 80, + rows: 24, + streamClient: { + onData: (data) => { + output += data + }, + onExit + } + }) + await vi.waitFor(() => expect(output).toContain('orca-cleanup-ready'), { timeout: 3000 }) + handle.resize(100, 30) + handle.write('roundtrip\n') + await vi.waitFor(() => expect(output).toContain('reply:roundtrip'), { timeout: 3000 }) + host.pauseProducer('native-io-failure') + expect(process.kill(native.pid, 0)).toBe(true) + const failure = vi.spyOn(native, operation).mockImplementation(() => { + throw new Error('injected I/O failure') + }) + if (operation === 'write') { + handle.write('ignored') + } else { + handle.resize(100, 30) + } + failure.mockRestore() + + await host.kill('native-io-failure', { immediate }) + await vi.waitFor(() => expect(onExit).toHaveBeenCalledOnce(), { timeout: 3000 }) + expect(host.listSessions()).toHaveLength(0) + expect(() => process.kill(native.pid, 0)).toThrow( + expect.objectContaining({ code: 'ESRCH' }) + ) + expect(() => fstatSync(fd)).toThrow(expect.objectContaining({ code: 'EBADF' })) + } finally { + // Only this test's still-owned native child is eligible for emergency cleanup. + if (!exited) { + native.kill('SIGKILL') + } + await vi.waitFor(() => expect(exited).toBe(true), { timeout: 3000 }) + await host.dispose() + } + } + }, + 15000 + ) +}) diff --git a/src/main/daemon/pty-subprocess/subprocess-handle.ts b/src/main/daemon/pty-subprocess/subprocess-handle.ts index e2abd59c6ea..974602dfe84 100644 --- a/src/main/daemon/pty-subprocess/subprocess-handle.ts +++ b/src/main/daemon/pty-subprocess/subprocess-handle.ts @@ -28,6 +28,8 @@ export function createDaemonPtySubprocessHandle(args: { const nativeProc = proc as DisposableNativePty const events = new PtyPreListenerEvents() let dead = false + // I/O failure is not exit evidence; keep termination and producer flow control available. + let ioFailed = false let disposed = false let nodePtyKillIssued = false const foreground = createPtyForegroundProcessTracker({ @@ -44,19 +46,18 @@ export function createDaemonPtySubprocessHandle(args: { events.acceptData(data) }) proc.onExit(({ exitCode, signal }) => { - events.acceptExit({ - exitCode, - signal, - hostReportsChildExitStatus: args.reportsChildExitStatus - }) - }) - proc.onExit(() => { + // Exit listeners may re-enter cleanup; retire signal authority before notifying them. dead = true foreground.markDead() // Why: neutralize kill synchronously so a later async socket-close SIGHUP cannot hit a recycled pid. if (process.platform !== 'win32') { nativeProc.kill = () => {} } + events.acceptExit({ + exitCode, + signal, + hostReportsChildExitStatus: args.reportsChildExitStatus + }) }) const slavePath = readPtySlavePath(proc) @@ -73,23 +74,23 @@ export function createDaemonPtySubprocessHandle(args: { confirmForegroundProcess: foreground.confirmForegroundProcess, confirmShellForeground: foreground.confirmShellForeground, write: (data) => { - if (dead) { + if (dead || ioFailed) { return } try { proc.write(data) } catch { - dead = true + ioFailed = true } }, resize: (cols, rows) => { - if (dead || !isValidPtySize(cols, rows)) { + if (dead || ioFailed || !isValidPtySize(cols, rows)) { return } try { proc.resize(cols, rows) } catch { - dead = true + ioFailed = true } }, // WindowsTerminal also wires _socket to the ConPTY conout pipe, so pausing backpressures the child. @@ -114,7 +115,7 @@ export function createDaemonPtySubprocessHandle(args: { } }, clear: () => { - if (dead) { + if (dead || ioFailed) { return } try { diff --git a/src/main/gitlab/mr-file-diffs.ts b/src/main/gitlab/mr-file-diffs.ts index 0e16567a294..8f6d10ce410 100644 --- a/src/main/gitlab/mr-file-diffs.ts +++ b/src/main/gitlab/mr-file-diffs.ts @@ -25,19 +25,23 @@ export function countDiffLines(diff: string): { additions: number; deletions: nu // diff line `---`, colliding with the `--- a/file` header — so it must // be counted once inside a hunk, not skipped. let inHunk = false - for (const line of diff.split('\n')) { - if (line.startsWith('@@')) { + let cursor = 0 + while (cursor < diff.length) { + if (diff.startsWith('@@', cursor)) { inHunk = true - continue + } else if (inHunk) { + const prefix = diff.charCodeAt(cursor) + if (prefix === 43) { + additions += 1 + } else if (prefix === 45) { + deletions += 1 + } } - if (!inHunk) { - continue - } - if (line.startsWith('+')) { - additions += 1 - } else if (line.startsWith('-')) { - deletions += 1 + const newline = diff.indexOf('\n', cursor) + if (newline === -1) { + break } + cursor = newline + 1 } return { additions, deletions } } diff --git a/src/main/gitlab/work-item-details.test.ts b/src/main/gitlab/work-item-details.test.ts index f1e2297e14b..9de6bb43d03 100644 --- a/src/main/gitlab/work-item-details.test.ts +++ b/src/main/gitlab/work-item-details.test.ts @@ -458,4 +458,42 @@ describe('countDiffLines', () => { // Why: the `@@` hunk check runs first, so it must not swallow `+`/`-` content. expect(countDiffLines('@@ -1 +1 @@\n-@@ old\n+@@ new')).toEqual({ additions: 1, deletions: 1 }) }) + + // Why: the scan now reads a prefix code unit at a byte cursor rather than a split + // segment, so line-ending and non-ASCII shapes are the new regression surface. + it('counts a CRLF hunk the same as an LF hunk', () => { + expect(countDiffLines('@@ -1 +1,2 @@\r\n-old\r\n+a\r\n+b\r\n')).toEqual({ + additions: 2, + deletions: 1 + }) + }) + + it('treats a lone CR as content, not a line break', () => { + expect(countDiffLines('@@ -1 +1 @@\n-old\r+new')).toEqual({ additions: 0, deletions: 1 }) + }) + + it('counts lines whose content is multi-byte or a surrogate pair', () => { + expect(countDiffLines('@@ -1 +1 @@\n-é ünïcode\n+🚀 rocket')).toEqual({ + additions: 1, + deletions: 1 + }) + }) + + it('ignores non-ASCII context lines and blank lines inside a hunk', () => { + expect(countDiffLines('@@ -1 +1 @@\n é leading accent\n 🚀 leading emoji\n\n')).toEqual({ + additions: 0, + deletions: 0 + }) + }) + + it('counts large diff prefixes without allocating a string array for every line', () => { + const diff = `--- a/file\n+++ b/file\n@@ -1 +1 @@\n${'-old\n+new\n context\n'.repeat(10000)}` + const split = vi.spyOn(String.prototype, 'split') + try { + expect(countDiffLines(diff)).toEqual({ additions: 10000, deletions: 10000 }) + expect(split.mock.calls.length).toBe(0) + } finally { + split.mockRestore() + } + }) }) diff --git a/src/main/ipc/filesystem-allowed-roots.test.ts b/src/main/ipc/filesystem-allowed-roots.test.ts index f94c99c5fdb..2ba6eaec367 100644 --- a/src/main/ipc/filesystem-allowed-roots.test.ts +++ b/src/main/ipc/filesystem-allowed-roots.test.ts @@ -8,6 +8,7 @@ import { listRepoWorktreeGraph } from '../repo-worktrees' import type * as ProjectGroupsModule from '../../shared/project-groups' import { buildProjectGroupChildIndex, getProjectGroupSubtreeIds } from '../../shared/project-groups' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' +import type * as CrossPlatformPathModule from '../../shared/cross-platform-path' import { getWorktreeMirrorDistro } from '../project-runtime-git-options' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ProjectGroup } from '../../shared/project-group-types' @@ -32,6 +33,13 @@ vi.mock('../../shared/project-groups', async () => { } }) +vi.mock('../../shared/cross-platform-path', async () => { + const actual = await vi.importActual( + '../../shared/cross-platform-path' + ) + return { ...actual, isPathInsideOrEqual: vi.fn(actual.isPathInsideOrEqual) } +}) + type StoreFixture = { repos: Repo[] projects: Project[] @@ -257,6 +265,42 @@ beforeEach(() => { }) describe('getAllowedRoots', () => { + it('stops scanning repositories when a local candidate settles each folder scope', () => { + const fixture: StoreFixture = { + repos: Array.from({ length: 1_000 }, (_, index) => + makeRepo({ id: `repo-${index}`, path: `/folders/root/repo-${index}` }) + ), + projects: [], + projectGroups: [], + folderWorkspaces: Array.from({ length: 100 }, (_, index) => + makeWorkspace({ id: `folder-${index}`, folderPath: '/folders/root' }) + ) + } + const { store } = makeCountingStore(fixture) + vi.mocked(isPathInsideOrEqual).mockClear() + const actual = getAllowedRoots(store) + expect(isPathInsideOrEqual).toHaveBeenCalledTimes(100) + vi.mocked(isPathInsideOrEqual).mockClear() + expect(actual).toEqual(referenceAllowedRoots(store)) + expect(isPathInsideOrEqual).toHaveBeenCalledTimes(100_000) + }) + + it('preserves empty, remote-only, mixed and explicit remote folder scopes in any repo order', () => { + const fixture = makeMixedFixture() + fixture.repos.push( + makeRepo({ + id: 'local-in-remote-group', + path: '/local/mixed', + projectGroupId: 'group-remote' + }) + ) + for (let index = 0; index < fixture.repos.length; index += 1) { + fixture.repos.push(fixture.repos.shift()!) + const { store } = makeCountingStore(fixture) + expect(getAllowedRoots(store)).toEqual(referenceAllowedRoots(store)) + } + }) + it('produces the same roots as the pre-change implementation', () => { const { store } = makeCountingStore(makeMixedFixture()) diff --git a/src/main/ipc/filesystem-allowed-roots.ts b/src/main/ipc/filesystem-allowed-roots.ts index cef249430c6..fb4e9854c2e 100644 --- a/src/main/ipc/filesystem-allowed-roots.ts +++ b/src/main/ipc/filesystem-allowed-roots.ts @@ -27,20 +27,6 @@ export function getLocalRepos(store: Store) { return filterLocalRepos(store.getRepos()) } -function getFolderScopeCandidateRepos( - folderPath: string, - projectGroupId: string, - childGroupIndex: ProjectGroupChildIndex, - repos: readonly Repo[] -): Repo[] { - const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId) - return repos.filter( - (repo) => - (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || - isPathInsideOrEqual(folderPath, repo.path) - ) -} - function isRemoteOnlyFolderScope( folderPath: string, projectGroupId: string, @@ -51,13 +37,21 @@ function isRemoteOnlyFolderScope( if (connectionId) { return true } - const candidates = getFolderScopeCandidateRepos( - folderPath, - projectGroupId, - childGroupIndex, - repos - ) - return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) + const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId) + let hasRemoteCandidate = false + for (const repo of repos) { + if ( + (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || + isPathInsideOrEqual(folderPath, repo.path) + ) { + // One local candidate settles the scope without scanning the remaining repositories. + if (!repo.connectionId) { + return false + } + hasRemoteCandidate = true + } + } + return hasRemoteCandidate } function getFolderWorkspaceConnectionId( diff --git a/src/main/ipc/pty-activation-inventory-scope.test.ts b/src/main/ipc/pty-activation-inventory-scope.test.ts new file mode 100644 index 00000000000..a6cb5dabad5 --- /dev/null +++ b/src/main/ipc/pty-activation-inventory-scope.test.ts @@ -0,0 +1,139 @@ +import { describe, expect, it, vi } from 'vitest' +import { setupPtyIpcSuite } from './pty-ipc-test-harness' +import { registerSshPtyProvider, getLocalPtyProvider } from './pty' +import { installPtyInspectIpcHandlers } from './pty/ipc/inspect' +import { ptyOwnership } from './pty/provider/ownership-state' + +vi.mock('electron', () => import('./pty-ipc-mock-registry').then((m) => m.electronModuleMock())) +vi.mock('fs', () => import('./pty-ipc-mock-registry').then((m) => m.fsModuleMock())) +vi.mock('node-pty', () => import('./pty-ipc-mock-registry').then((m) => m.nodePtyModuleMock())) +vi.mock('node:child_process', async (importOriginal) => + (await import('./pty-ipc-mock-registry')).childProcessModuleMock(await importOriginal()) +) +vi.mock('../opencode/hook-service', () => + import('./pty-ipc-mock-registry').then((m) => m.openCodeHookServiceModuleMock()) +) +vi.mock('../mimo/hook-service', () => + import('./pty-ipc-mock-registry').then((m) => m.mimoHookServiceModuleMock()) +) +vi.mock('../agent-hooks/server', () => + import('./pty-ipc-mock-registry').then((m) => m.agentHookServerModuleMock()) +) +vi.mock('../pi/titlebar-extension-service', () => + import('./pty-ipc-mock-registry').then((m) => m.piTitlebarExtensionModuleMock()) +) +vi.mock('../pwsh', () => import('./pty-ipc-mock-registry').then((m) => m.pwshModuleMock())) +vi.mock('../wsl', async (importOriginal) => + (await import('./pty-ipc-mock-registry')).wslModuleMock(await importOriginal()) +) +vi.mock('../telemetry/client', () => + import('./pty-ipc-mock-registry').then((m) => m.telemetryClientModuleMock()) +) +vi.mock('../telemetry/classify-error', () => + import('./pty-ipc-mock-registry').then((m) => m.classifyErrorModuleMock()) +) +vi.mock('../cli/linux-terminal-orca-cli-shim', () => + import('./pty-ipc-mock-registry').then((m) => m.linuxCliShimModuleMock()) +) +vi.mock('../memory/pty-registry', () => + import('./pty-ipc-mock-registry').then((m) => m.ptyRegistryModuleMock()) +) +vi.mock('../agent-hooks/migration-unsupported-pty-state', () => + import('./pty-ipc-mock-registry').then((m) => m.migrationUnsupportedPtyModuleMock()) +) +vi.mock('../codex/codex-pane-account-registry', () => + import('./pty-ipc-mock-registry').then((m) => m.codexPaneAccountRegistryModuleMock()) +) +vi.mock('../codex/codex-state-db-backfill-recovery', () => + import('./pty-ipc-mock-registry').then((m) => m.codexBackfillRecoveryModuleMock()) +) + +describe('scoped activation PTY inventory', () => { + const { handlers, installDaemonTestProvider } = setupPtyIpcSuite() + + function install() { + const localList = vi.fn(async () => [{ id: 'local', cwd: '/', title: 'shell' }]) + installDaemonTestProvider({ listProcesses: localList }) + const remoteLists = Array.from({ length: 50 }, (_, index) => { + const list = vi.fn(async () => [ + { + id: `ssh:host-${index}@@pty-1`, + cwd: '/remote', + title: 'agent', + worktreeId: 'repo::/remote' + } + ]) + registerSshPtyProvider(`host-${index}`, { + ...getLocalPtyProvider(), + listProcesses: list, + providesAgentSessionOwnerListings: () => true + }) + return list + }) + const startup = vi.fn(async () => {}) + installPtyInspectIpcHandlers({ getLocalPtyProviderStartupPromise: startup }) + const list = (scope?: unknown) => handlers.get('pty:listSessions')!(null, scope) + return { localList, remoteLists, startup, list } + } + + it('queries only the chosen SSH provider and preserves workspace and ownership evidence', async () => { + const { list, localList, remoteLists, startup } = install() + expect(await list({ connectionId: 'host-17' })).toEqual([ + { + id: 'ssh:host-17@@pty-1', + cwd: '/remote', + title: 'agent', + worktreeId: 'repo::/remote', + agentOwnership: 'absent' + } + ]) + expect(remoteLists[17]).toHaveBeenCalledOnce() + expect(remoteLists.reduce((count, mock) => count + mock.mock.calls.length, 0)).toBe(1) + expect(localList).not.toHaveBeenCalled() + expect(startup).not.toHaveBeenCalled() + expect(ptyOwnership.get('ssh:host-17@@pty-1')).toBe('host-17') + }) + + it('waits for local startup and never visits remote providers for a local scope', async () => { + const { list, localList, remoteLists, startup } = install() + let release!: () => void + startup.mockImplementation( + () => + new Promise((resolve) => { + release = resolve + }) + ) + const pending = list({ connectionId: null }) + expect(localList).not.toHaveBeenCalled() + release() + await pending + expect(localList).toHaveBeenCalledOnce() + expect(remoteLists.every((mock) => mock.mock.calls.length === 0)).toBe(true) + }) + + it('propagates selected-host failure and never substitutes the local inventory', async () => { + const { list, localList, remoteLists } = install() + remoteLists[3].mockRejectedValue(new Error('relay unavailable')) + await expect(list({ connectionId: 'host-3' })).rejects.toThrow('relay unavailable') + await expect(list({ connectionId: 'missing' })).rejects.toThrow('No PTY provider') + expect(localList).not.toHaveBeenCalled() + }) + + it.each([null, {}, { connectionId: '' }, { connectionId: 42 }])( + 'rejects malformed scope %j before inventory admission', + async (scope) => { + const { list, localList, remoteLists } = install() + await expect(list(scope)).rejects.toThrow('invalid_pty_session_list_scope') + expect(localList).not.toHaveBeenCalled() + expect(remoteLists.every((mock) => mock.mock.calls.length === 0)).toBe(true) + } + ) + + it('preserves unscoped diagnostic inventory and its remote-error fallback', async () => { + const { list, localList, remoteLists } = install() + remoteLists[3].mockRejectedValue(new Error('relay unavailable')) + expect(await list()).toHaveLength(50) + expect(localList).toHaveBeenCalledOnce() + expect(remoteLists.every((mock) => mock.mock.calls.length === 1)).toBe(true) + }) +}) diff --git a/src/main/ipc/pty/ipc/inspect.ts b/src/main/ipc/pty/ipc/inspect.ts index 5a6d03d8855..37226017b8f 100644 --- a/src/main/ipc/pty/ipc/inspect.ts +++ b/src/main/ipc/pty/ipc/inspect.ts @@ -6,10 +6,11 @@ import { PtyProcessListAdmission, visitPtyProcessListingsInBatches } from '../../../providers/pty-process-list-admission' -import type { PtyListedSession } from '../../../../shared/pty-listed-session' +import type { PtyListedSession, PtySessionListScope } from '../../../../shared/pty-listed-session' import { ptyOwnership } from '../provider/ownership-state' import { getProviderForPty, + getProvider, hasPtyProviderForInspection, registeredPtyProviders, sshProviders, @@ -40,37 +41,58 @@ export function installPtyInspectIpcHandlers(deps: { ) } - ipcMain.handle('pty:listSessions', async (): Promise => { - const deduped = new Map() - const admission = new PtyProcessListAdmission() - await visitPtyProcessListingsInBatches( - registeredPtyProviders(), - ({ provider, connectionId }) => - connectionId === null ? provider.listProcesses() : provider.listProcesses().catch(() => []), - ({ provider, connectionId }, sessions) => { - for (const rawSession of sessions) { - const session = admission.admit(rawSession) - // Why: kill actions only send back the PTY id, so rebuild ownership while listing to keep reconnect-discovered remote sessions routed to their provider. - ptyOwnership.set(session.id, connectionId) - deduped.set(session.id, { - id: session.id, - cwd: session.cwd, - title: session.title, - // Why: the renderer's binding map is empty during restore, so ownership is the only - // liveness evidence it has. Absence is authoritative only from a provider that - // serializes claims — otherwise it is 'unknown', never 'absent' (#8459). - agentOwnership: - (session.agentSessionOwners?.length ?? 0) > 0 - ? 'present' - : provider.providesAgentSessionOwnerListings?.(session.id) === true - ? 'absent' - : 'unknown' - }) + ipcMain.handle( + 'pty:listSessions', + async (_event, scope?: PtySessionListScope): Promise => { + if (scope !== undefined) { + if ( + !scope || + (scope.connectionId !== null && + (typeof scope.connectionId !== 'string' || !scope.connectionId.trim())) + ) { + throw new Error('invalid_pty_session_list_scope') + } + // Select the daemon only after startup has handed off ownership. + if (scope.connectionId === null) { + await getLocalPtyProviderStartupPromise() } } - ) - return Array.from(deduped.values()) - }) + const deduped = new Map() + const admission = new PtyProcessListAdmission() + await visitPtyProcessListingsInBatches( + scope === undefined + ? registeredPtyProviders() + : [{ provider: getProvider(scope.connectionId), connectionId: scope.connectionId }], + ({ provider, connectionId }) => + connectionId === null || scope !== undefined + ? provider.listProcesses() + : provider.listProcesses().catch(() => []), + ({ provider, connectionId }, sessions) => { + for (const rawSession of sessions) { + const session = admission.admit(rawSession) + // Why: kill actions only send back the PTY id, so rebuild ownership while listing to keep reconnect-discovered remote sessions routed to their provider. + ptyOwnership.set(session.id, connectionId) + deduped.set(session.id, { + id: session.id, + cwd: session.cwd, + title: session.title, + ...(session.worktreeId !== undefined ? { worktreeId: session.worktreeId } : {}), + // Why: the renderer's binding map is empty during restore, so ownership is the only + // liveness evidence it has. Absence is authoritative only from a provider that + // serializes claims — otherwise it is 'unknown', never 'absent' (#8459). + agentOwnership: + (session.agentSessionOwners?.length ?? 0) > 0 + ? 'present' + : provider.providesAgentSessionOwnerListings?.(session.id) === true + ? 'absent' + : 'unknown' + }) + } + } + ) + return Array.from(deduped.values()) + } + ) ipcMain.handle( 'pty:getAuthoritativeBufferSnapshotCapabilities', diff --git a/src/main/ipc/pty/pane/stable-pane-relay-absence-respawn.test.ts b/src/main/ipc/pty/pane/stable-pane-relay-absence-respawn.test.ts index 9a77dd7fbb9..e68bcf6ec99 100644 --- a/src/main/ipc/pty/pane/stable-pane-relay-absence-respawn.test.ts +++ b/src/main/ipc/pty/pane/stable-pane-relay-absence-respawn.test.ts @@ -81,6 +81,49 @@ function sessionStore(leaves: string[]): { store: Store; read: () => WorkspaceSe } describe('stable pane adoption after the relay reports the PTY absent', () => { + it.each([false, true])( + 'reattaches a live pane without launching a provider process (settled worker: %s)', + async (settledWorker) => { + const { store, read } = sessionStore([LEAF]) + const paneKey = `${OWNER.tabId}:${LEAF}` + const record = { + paneKey, + tabId: OWNER.tabId, + worktreeId: WORKTREE, + agent: 'claude' as const, + providerSession: { key: 'session_id' as const, id: 'provider-session' }, + prompt: '', + state: 'done' as const, + capturedAt: 1, + updatedAt: 1, + ...(settledWorker ? { automaticResumeBlockedBy: 'legacy-orchestration-worker' } : {}) + } + store.setWorkspaceSession({ + ...read(), + sleepingAgentSessionsByPaneKey: { [paneKey]: record } + }) + const spawn = vi.fn().mockResolvedValue({ id: OWNER.ptyId, isReattach: true }) + const onFreshSpawn = vi.fn() + const result = await spawnForStablePane({ + runtime: undefined, + store, + worktreeId: WORKTREE, + provider: { spawn } as unknown as IPtyProvider, + spawnOptions: { cols: 80, rows: 24, command: 'claude --resume provider-session' }, + owner: OWNER, + connectionId: 'conn-1', + resolveOwner: () => OWNER, + onFreshSpawn + }) + expect(result.owner).toBe(OWNER) + expect(spawn).toHaveBeenCalledExactlyOnceWith( + expect.objectContaining({ sessionId: OWNER.ptyId, attachOnly: true, command: undefined }) + ) + expect(onFreshSpawn).not.toHaveBeenCalled() + expect(read().tabsByWorktree[WORKTREE]).toHaveLength(1) + } + ) + it('spawns fresh once the relay has positively answered for that id', async () => { const { run, spawn } = spawnAfterAttachRejection( new SshPtyAbsentFromRelayError(`${SSH_SESSION_EXPIRED_ERROR}: pty-1`) diff --git a/src/main/ipc/runtime.ts b/src/main/ipc/runtime.ts index 3901d8b1ffa..6237b8d040d 100644 --- a/src/main/ipc/runtime.ts +++ b/src/main/ipc/runtime.ts @@ -11,6 +11,7 @@ import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' import { TERMINAL_FIT_RESTORE_DEADLINE_MS } from '../../shared/terminal-fit-restore-deadline' import { + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' @@ -80,6 +81,7 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientKind: 'runtime', connectionId: desktopSenders.connectionIdFor(event.sender), clientCapabilities: [ + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ] @@ -128,6 +130,7 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientKind: 'runtime', connectionId, clientCapabilities: [ + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ] diff --git a/src/main/jira/jira-issue-search.ts b/src/main/jira/jira-issue-search.ts index 743342f7727..0d39c841ed4 100644 --- a/src/main/jira/jira-issue-search.ts +++ b/src/main/jira/jira-issue-search.ts @@ -1,3 +1,4 @@ +import { sortByUpdatedAtDescending } from '../../shared/updated-at-order' import type { JiraIssue, JiraIssueFilter, JiraSiteSelection } from '../../shared/jira-types' import { acquire, release } from './request-queue' import { apiBasePath, jiraRequest, type JiraClientForSite } from './authenticated-request' @@ -18,9 +19,7 @@ function clampLimit(limit: number | undefined, fallback = 30): number { } function sortAndLimitIssues(issues: JiraIssue[], limit: number): JiraIssue[] { - return issues - .sort((a, b) => new Date(b.updatedAt).getTime() - new Date(a.updatedAt).getTime()) - .slice(0, limit) + return sortByUpdatedAtDescending(issues).slice(0, limit) } function filterToJql(filter: JiraIssueFilter): string { diff --git a/src/main/linear/linear-issue-query-support.ts b/src/main/linear/linear-issue-query-support.ts index 216fb5f9b80..3154284ae2a 100644 --- a/src/main/linear/linear-issue-query-support.ts +++ b/src/main/linear/linear-issue-query-support.ts @@ -1,3 +1,4 @@ +import { sortByUpdatedAtDescending } from '../../shared/updated-at-order' import type { LinearIssue } from '../../shared/linear/issue-types' import type { LinearWorkspaceSelection } from '../../shared/linear/workspace-types' import { LINEAR_ISSUE_API_PAGE_SIZE_MAX } from '../../shared/linear/issue-read-limits' @@ -30,18 +31,14 @@ export async function mapIssueForWorkspace( } export function sortAndLimitIssues(issues: LinearIssue[], limit: number): LinearIssue[] { - return issues - .sort((a, b) => new Date(b.updatedAt).getTime() - new Date(a.updatedAt).getTime()) - .slice(0, limit) + return sortByUpdatedAtDescending(issues).slice(0, limit) } export function sortLimitAndDescribeIssues( issues: LinearIssue[], limit: number ): { items: LinearIssue[]; clipped: boolean } { - const sorted = issues.sort( - (a, b) => new Date(b.updatedAt).getTime() - new Date(a.updatedAt).getTime() - ) + const sorted = sortByUpdatedAtDescending(issues) return { items: sorted.slice(0, limit), clipped: sorted.length > limit diff --git a/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts b/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts index ed86d81861e..13673626ebc 100644 --- a/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts @@ -432,6 +432,86 @@ describe('payload bounds on import', () => { expect(body.output.byteLength).toBe(64 * 1024) expect(body.output.head).toHaveLength(1_024) }) + + it('bounds an imported subagent roster by entry count, label and id', async () => { + // The import reads an untrusted file: nothing upstream bounded either string. + const oversized = 'z'.repeat(20 * 1024) + const journal = await open('claude', CLAUDE_SESSION) + await appendLegacyTranscriptMessages({ + journal, + agent: 'claude', + sessionId: CLAUDE_SESSION, + fence: 1, + messages: [ + { + id: 'legacy-roster', + role: 'assistant', + timestamp: null, + source: 'transcript', + blocks: [ + { + type: 'subagent-group', + groupId: 'group-1', + agents: Array.from({ length: 80 }, (_, index) => ({ + id: index === 0 ? oversized : `task-${index}`, + label: index === 0 ? oversized : `label-${index}`, + state: 'working' as const + })) + } + ] + } + ] + }) + + const body = journal.snapshot().items[0]?.body + const block = body?.kind === 'message' ? body.blocks[0] : undefined + if (block?.type !== 'subagent-group') { + throw new Error('expected a subagent-group block') + } + expect(block.agents).toHaveLength(64) + expect(block.agents[0]?.label.length).toBeLessThan(oversized.length) + expect(block.agents[0]?.id.length).toBeLessThan(oversized.length) + expect(block.agents[0]?.id.startsWith('z')).toBe(true) + }) + + it('bounds a roster id in the shared format, keeping a shared prefix distinct', async () => { + // The id is the roster key: it takes the same bounded-id format the wires + // use, so a later wire bound is a no-op instead of a second, merging clip. + const head = 'y'.repeat(512) + const journal = await open('claude', CLAUDE_SESSION) + await appendLegacyTranscriptMessages({ + journal, + agent: 'claude', + sessionId: CLAUDE_SESSION, + fence: 1, + messages: [ + { + id: 'legacy-roster-collision', + role: 'assistant', + timestamp: null, + source: 'transcript', + blocks: [ + { + type: 'subagent-group', + groupId: 'group-1', + agents: [ + { id: `${head}-one`, label: 'Audit', state: 'working' as const }, + { id: `${head}-two`, label: 'Audit', state: 'working' as const } + ] + } + ] + } + ] + }) + + const body = journal.snapshot().items[0]?.body + const block = body?.kind === 'message' ? body.blocks[0] : undefined + if (block?.type !== 'subagent-group') { + throw new Error('expected a subagent-group block') + } + expect(block.agents[0]?.id).not.toBe(block.agents[1]?.id) + expect(block.agents[0]?.id).toHaveLength(512) + }) }) describe('import failures', () => { diff --git a/src/main/native-chat/agent-session-journal/journal-legacy-import.ts b/src/main/native-chat/agent-session-journal/journal-legacy-import.ts index 0907ebd28f2..801c91fb534 100644 --- a/src/main/native-chat/agent-session-journal/journal-legacy-import.ts +++ b/src/main/native-chat/agent-session-journal/journal-legacy-import.ts @@ -28,6 +28,7 @@ import { decodeOmpTranscriptLine } from '../transcript-line-decoders' import { decodeTranscriptStream } from '../transcript-stream-lines' +import { boundSubagentEntryId } from '../subagent-entry-id-bounds' import { createLegacyIdentityTracker } from './journal-legacy-identity' import type { JournalReplacementItem } from './journal-epoch-replacement' import { @@ -47,6 +48,8 @@ export type LegacyImportOptions = ResolveSessionFileOptions & { } const MAX_LEGACY_IMPORT_SOURCE_BYTES = 16 * 1024 * 1024 +/** A roster is a status list; an imported one is as untrusted as any other block. */ +const MAX_LEGACY_IMPORT_SUBAGENTS = 64 export type LegacyImportResult = | { @@ -271,5 +274,17 @@ function boundBlock(block: NativeChatBlock, limits: JournalPayloadLimits): Nativ if (block.type === 'tool-call') { return { ...block, input: boundToolInput(block.input, limits) } } + if (block.type === 'subagent-group') { + return { + ...block, + agents: block.agents.slice(0, MAX_LEGACY_IMPORT_SUBAGENTS).map((agent) => ({ + ...agent, + // The id is the roster key, so it is bounded with a digest rather than + // clipped to a prefix that two distinct children could share. + id: boundSubagentEntryId(agent.id), + label: boundInlineText(agent.label, limits).text + })) + } + } return block } diff --git a/src/main/native-chat/agent-session-journal/journal-open.ts b/src/main/native-chat/agent-session-journal/journal-open.ts index 350d2e9cfb9..e4cc5e0f73e 100644 --- a/src/main/native-chat/agent-session-journal/journal-open.ts +++ b/src/main/native-chat/agent-session-journal/journal-open.ts @@ -167,10 +167,11 @@ export function readJournalRowsAfterCursor( db: Database.Database, sessionId: string, epoch: string, - afterSequence: number + afterSequence: number, + limit?: number ): JournalRow[] { const rows: JournalRow[] = [] - for (const stored of readJournalRowsAfter(db, sessionId, epoch, afterSequence)) { + for (const stored of readJournalRowsAfter(db, sessionId, epoch, afterSequence, limit)) { const parsed = parseJournalRow(stored.rowJson) if (!parsed.ok) { break diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts index 2dc0ac1aeb2..255e4ed184c 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts @@ -5,7 +5,10 @@ import { boundJournalKeyComponent, MAX_JOURNAL_KEY_COMPONENT_CHARS } from '../../../shared/agent-session-journal-item-key' -import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' +import type { + AgentJournalItemIdentity, + AgentJournalMessageItem +} from '../../../shared/agent-session-journal-types' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' import { applyJournalRow, @@ -14,6 +17,7 @@ import { renderJournalState, type JournalReducerState } from './journal-reducer' +import { buildJournalItemRow, buildJournalTombstoneRow } from './journal-row-builders' import type { JournalRow } from './journal-row-schema' const EPOCH = 'epoch-1' @@ -444,3 +448,49 @@ describe('bounded item-key collisions', () => { ]) }) }) + +describe('re-adding a tombstoned row', () => { + it('builds the rebuilt row above the tombstone that removed it', () => { + const identity: AgentJournalItemIdentity = { provider: 'orca', clientMessageId: 'roster' } + const itemId = agentJournalItemKey(identity) + const state = createJournalReducerState('session-1', EPOCH) + applyJournalRow( + state, + buildJournalItemRow({ state, identity, body: text('first'), seq: 1, fence: 1, ts: 1_001 }) + ) + applyJournalRow(state, buildJournalTombstoneRow({ state, itemId, seq: 2, fence: 1, ts: 1_002 })) + expect(renderJournalState(state).items).toEqual([]) + + // Same identity, re-added later in the session: a revision built only from + // `items` would restart at 1 and lose to the tombstone forever. + applyJournalRow( + state, + buildJournalItemRow({ state, identity, body: text('second'), seq: 3, fence: 1, ts: 1_003 }) + ) + expect(renderJournalState(state).items.map((item) => item.body)).toEqual([text('second')]) + }) + + // `upsertItem` clearing the tombstone on a re-add is a map-state invariant: + // `items` and `tombstones` stay disjoint, so a re-added row is never both + // present and removed. Revision ordering is now independent of it — + // `buildJournalTombstoneRow` takes `max(itemRevision, tombstoneRevision) + 1` + // — so what this pins is the map state itself, not the ranking. + it('removes the row again after it was re-added', () => { + const identity: AgentJournalItemIdentity = { provider: 'orca', clientMessageId: 'roster' } + const itemId = agentJournalItemKey(identity) + const state = createJournalReducerState('session-1', EPOCH) + applyJournalRow( + state, + buildJournalItemRow({ state, identity, body: text('first'), seq: 1, fence: 1, ts: 1_001 }) + ) + applyJournalRow(state, buildJournalTombstoneRow({ state, itemId, seq: 2, fence: 1, ts: 1_002 })) + applyJournalRow( + state, + buildJournalItemRow({ state, identity, body: text('second'), seq: 3, fence: 1, ts: 1_003 }) + ) + expect(state.tombstones.get(itemId)).toBeUndefined() + + applyJournalRow(state, buildJournalTombstoneRow({ state, itemId, seq: 4, fence: 1, ts: 1_004 })) + expect(renderJournalState(state).items).toEqual([]) + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-row-builders.ts b/src/main/native-chat/agent-session-journal/journal-row-builders.ts index 5c77c180c68..db82e86318f 100644 --- a/src/main/native-chat/agent-session-journal/journal-row-builders.ts +++ b/src/main/native-chat/agent-session-journal/journal-row-builders.ts @@ -148,7 +148,13 @@ export function buildJournalItemRow(input: { }): JournalItemRow { const itemId = agentJournalItemKey(input.identity) const resolved = input.state.aliases.get(itemId) ?? itemId - const revision = (input.state.items.get(resolved)?.revision ?? 0) + 1 + // A tombstoned row keeps its revision in `tombstones`, and the reducer drops + // any item at or below it — so a re-add has to outrank the tombstone too. + const revision = + Math.max( + input.state.items.get(resolved)?.revision ?? 0, + input.state.tombstones.get(resolved) ?? 0 + ) + 1 return { kind: 'item', itemId, @@ -170,7 +176,15 @@ export function buildJournalTombstoneRow(input: { return { kind: 'tombstone', itemId: input.itemId, - revision: (input.state.items.get(resolved)?.revision ?? 0) + 1, + // Symmetric with the item builder: `upsertItem` clearing the tombstone on a + // re-add is what keeps the two maps disjoint, and that invariant lives in the + // reducer. Outranking both here means a repeat removal cannot be dropped as a + // stale revision if it ever stops holding. + revision: + Math.max( + input.state.items.get(resolved)?.revision ?? 0, + input.state.tombstones.get(resolved) ?? 0 + ) + 1, ...journalRowBase(input.state.epoch, input.seq, input.fence, input.ts) } } diff --git a/src/main/native-chat/agent-session-journal/journal-row-table.ts b/src/main/native-chat/agent-session-journal/journal-row-table.ts index 306b3b300f2..a6689f2040a 100644 --- a/src/main/native-chat/agent-session-journal/journal-row-table.ts +++ b/src/main/native-chat/agent-session-journal/journal-row-table.ts @@ -20,6 +20,7 @@ const SELECT_EPOCH_ROWS = `SELECT epoch, seq, ts, row_json FROM journal_rows WHERE session_id = ? AND epoch = ? ORDER BY seq ASC` const SELECT_ROWS_AFTER = `SELECT epoch, seq, ts, row_json FROM journal_rows WHERE session_id = ? AND epoch = ? AND seq > ? ORDER BY seq ASC` +const SELECT_ROWS_AFTER_LIMITED = `${SELECT_ROWS_AFTER} LIMIT ?` const DELETE_SUFFIX = 'DELETE FROM journal_rows WHERE session_id = ? AND epoch = ? AND seq >= ?' export function readJournalSessionEpoch(db: Database.Database, sessionId: string): string | null { @@ -58,8 +59,14 @@ export function readJournalRowsAfter( db: Database.Database, sessionId: string, epoch: string, - afterSeq: number + afterSeq: number, + limit?: number ): JournalStoredRow[] { + if (limit !== undefined) { + return toStoredRows( + db.prepare(SELECT_ROWS_AFTER_LIMITED).all(sessionId, epoch, afterSeq, limit) + ) + } return toStoredRows(db.prepare(SELECT_ROWS_AFTER).all(sessionId, epoch, afterSeq)) } diff --git a/src/main/native-chat/agent-session-journal/journal-store.ts b/src/main/native-chat/agent-session-journal/journal-store.ts index e2936b2553d..3c16800099b 100644 --- a/src/main/native-chat/agent-session-journal/journal-store.ts +++ b/src/main/native-chat/agent-session-journal/journal-store.ts @@ -177,7 +177,7 @@ export class AgentSessionJournal { canonicalItemId = (itemId: string): string => resolveJournalItemId(this.state, itemId) - readSince(cursor: AgentJournalCursor): JournalReadSince { + readSince(cursor: AgentJournalCursor, limit?: number): JournalReadSince { return readJournalSince( { state: this.state, @@ -186,7 +186,8 @@ export class AgentSessionJournal { this.requireDatabase().db, this.identity.sessionId, this.state.epoch, - afterSequence + afterSequence, + limit ), readOnly: this.readOnly }, diff --git a/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer-protected.test.ts b/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer-protected.test.ts new file mode 100644 index 00000000000..ff82a8f63ed --- /dev/null +++ b/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer-protected.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest' +import { createAgentSessionDeltaCoalescer } from './agent-session-delta-coalescer' + +describe('protected provider streams', () => { + it('preserves protected prefixes beyond the ordinary count cap and evicts only ordinary streams', () => { + const protectedKeys = new Set() + const coalescer = createAgentSessionDeltaCoalescer({ + maxStreams: 1, + isProtected: (key) => protectedKeys.has(key), + emit: () => true, + schedule: () => () => {} + }) + // A typed start can arrive after the first delta, before the next output. + coalescer.append('command-0', 'before') + protectedKeys.add('command-0') + for (let index = 1; index < 448; index += 1) { + const key = `command-${index}` + protectedKeys.add(key) + expect(coalescer.append(key, 'before')).toBe(true) + } + coalescer.append('ordinary-1', 'temporary') + coalescer.append('ordinary-2', 'latest') + expect(coalescer.snapshot('ordinary-1')).toBeNull() + expect(coalescer.snapshot('ordinary-2')?.text).toBe('latest') + for (const key of protectedKeys) { + coalescer.append(key, 'after') + expect(coalescer.snapshot(key)?.text).toBe('beforeafter') + coalescer.forget(key) + expect(coalescer.snapshot(key)).toBeNull() + } + coalescer.dispose() + }) + + it('keeps aggregate and per-stream output byte ceilings under protected-key pressure', () => { + const coalescer = createAgentSessionDeltaCoalescer({ + maxStreams: 1, + maxRetainedBytes: 80, + maxTotalRetainedBytes: 120, + isProtected: () => true, + emit: () => true, + schedule: () => () => {} + }) + coalescer.append('first', 'a'.repeat(80)) + coalescer.append('second', 'b'.repeat(80)) + coalescer.append('third', 'c'.repeat(80)) + const snapshots = ['first', 'second', 'third'].map((key) => coalescer.snapshot(key)!) + expect(snapshots[0].text).toBe('a'.repeat(80)) + expect(snapshots.map((snapshot) => snapshot.truncated)).toEqual([false, true, true]) + expect( + snapshots.reduce((total, snapshot) => total + Buffer.byteLength(snapshot.text), 0) + ).toBeLessThanOrEqual(120) + coalescer.forget('first') + coalescer.append('new', 'd'.repeat(80)) + expect(coalescer.snapshot('new')?.text).toBe('d'.repeat(80)) + coalescer.dispose() + }) +}) diff --git a/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer.ts b/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer.ts index dbc63154b72..73dc08cc05c 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-delta-coalescer.ts @@ -37,6 +37,8 @@ export type AgentSessionDeltaCoalescerDeps = { maxTotalRetainedBytes?: number /** Maximum distinct item streams retained at once. */ maxStreams?: number + /** The caller byte-bounds protected metadata; only ordinary streams use the count cap. */ + isProtected?: (key: string) => boolean /** Injected by tests so a window can be driven without real time. */ schedule?: (run: () => void, ms: number) => () => void } @@ -83,8 +85,7 @@ export function createAgentSessionDeltaCoalescer( >() let totalRetainedBytes = 0 let cancelTimer: (() => void) | null = null - const streamOrder = new Map() - let nextOrder = 0 + const evictable = new Set() const flushKey = (key: string): boolean => { const stream = streams.get(key) @@ -130,9 +131,13 @@ export function createAgentSessionDeltaCoalescer( if (!stream) { // Evict the oldest stream before admitting a new attacker-controlled // id. Flush first so the retained prefix is durably visible. - if (streams.size >= maxStreams) { - const oldest = [...streamOrder.entries()].sort((a, b) => a[1] - b[1])[0]?.[0] + while (!deps.isProtected?.(key) && evictable.size >= maxStreams) { + const oldest = evictable.values().next().value if (oldest) { + if (deps.isProtected?.(oldest)) { + evictable.delete(oldest) + continue + } // Under sink backpressure the oldest stream must remain available // for a later retry; dropping it would lose already-observed output. if (!flushKey(oldest)) { @@ -143,8 +148,9 @@ export function createAgentSessionDeltaCoalescer( totalRetainedBytes -= evicted.retainedBytes } streams.delete(oldest) - streamOrder.delete(oldest) + evictable.delete(oldest) } + break } stream = { chunks: [], @@ -153,7 +159,11 @@ export function createAgentSessionDeltaCoalescer( truncated: false, dirty: false } - streamOrder.set(key, nextOrder++) + if (!deps.isProtected?.(key)) { + evictable.add(key) + } + } else if (deps.isProtected?.(key)) { + evictable.delete(key) } stream.observedBytes += Buffer.byteLength(delta, 'utf8') if (!stream.truncated) { @@ -184,14 +194,14 @@ export function createAgentSessionDeltaCoalescer( if (stream) { totalRetainedBytes -= stream.retainedBytes streams.delete(key) - streamOrder.delete(key) + evictable.delete(key) } }, dispose: () => { cancelTimer?.() cancelTimer = null streams.clear() - streamOrder.clear() + evictable.clear() totalRetainedBytes = 0 }, snapshot: (key) => { diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts b/src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts new file mode 100644 index 00000000000..5a9a6f20801 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/agent-session-history-forward-read-budget.test.ts @@ -0,0 +1,228 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import Database from '../../sqlite/sync-database' +import { + AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + type AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import type { AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { + insertJournalRow, + upsertJournalSessionRow +} from '../agent-session-journal/journal-row-table' +import * as journalReducer from '../agent-session-journal/journal-reducer' +import * as rowSchema from '../agent-session-journal/journal-row-schema' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { AgentSessionSubscribers } from './structured-agent-session-subscribers' +import { readAgentSessionHistory } from './agent-session-history-page' + +const identity: AgentSessionJournalIdentity = { + sessionId: 'bounded-catch-up', + workspaceId: 'folder-workspace', + hostId: 'remote-host', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} +const journals = createTrackedJournalOpener() +let root: string | undefined + +afterEach(async () => { + vi.restoreAllMocks() + await journals.closeAll() + if (root) { + await rm(root, { recursive: true, force: true }) + } +}) + +async function seedJournal(count: number) { + root = await mkdtemp(join(tmpdir(), 'orca-history-read-budget-')) + const { db } = openJournalDatabase(journalDatabaseFile(root)) + const base = { + v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + epoch: 'epoch-1', + fence: 1, + ts: 1_000 + } + try { + db.exec('BEGIN IMMEDIATE') + upsertJournalSessionRow(db, identity.sessionId, base.epoch, base.ts) + insertJournalRow(db, identity.sessionId, { + ...base, + kind: 'epoch', + seq: 1, + reason: 'session_created', + providerHandle: identity.providerHandle + }) + for (let index = 0; index < count; index += 1) { + insertJournalRow(db, identity.sessionId, { + ...base, + kind: 'item', + seq: index + 2, + itemId: `item-${index}`, + revision: 1, + body: { + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: `${index}:${'x'.repeat(4096)}` }] + } + }) + } + db.exec('COMMIT') + } finally { + db.close() + } + return journals.open({ identity, journalDir: root }) +} + +function observeForwardReads() { + const returnedRows: number[] = [] + const observed = new WeakSet() + const prepare = Database.prototype.prepare + vi.spyOn(Database.prototype, 'prepare').mockImplementation(function (this: Database, sql) { + const statement = prepare.call(this, sql) + if (sql.includes('seq > ?') && !observed.has(statement)) { + observed.add(statement) + const all = statement.all.bind(statement) + vi.spyOn(statement, 'all').mockImplementation((...args) => { + const rows = all(...args) + returnedRows.push(rows.length) + return rows + }) + } + return statement + }) + const parse = vi.spyOn(rowSchema, 'parseJournalRow') + return { returnedRows, parse } +} + +describe('forward history SQL read budget', () => { + it('reconnects through every page with one lookahead row per page', async () => { + const count = 2_000 + const journal = await seedJournal(count) + const { returnedRows, parse } = observeForwardReads() + const events: AgentSessionSubscribeEvent[] = [] + new AgentSessionSubscribers().open({ + id: 'reader', + sessionId: identity.sessionId, + journal, + fence: 1, + cursor: { epoch: journal.epoch, sequence: 1 }, + emit: (event) => events.push(event) + }) + const batches = events.filter((event) => event.type === 'batch') + expect(batches.flatMap((event) => event.batch.items.map((item) => item.itemId))).toEqual( + Array.from({ length: count }, (_, index) => `item-${index}`) + ) + expect(batches.at(-1)?.batch.cursor).toEqual(journal.cursor()) + expect(returnedRows).toEqual([...Array(9).fill(201), 200]) + expect(parse).toHaveBeenCalledTimes(2_009) + }) + + it('reduces the timeline once for the whole catch-up, not once per page', async () => { + const journal = await seedJournal(2_000) + const render = vi.spyOn(journalReducer, 'renderJournalState') + const events: AgentSessionSubscribeEvent[] = [] + new AgentSessionSubscribers().open({ + id: 'reader', + sessionId: identity.sessionId, + journal, + fence: 1, + cursor: { epoch: journal.epoch, sequence: 1 }, + emit: (event) => events.push(event) + }) + // Catch-up is synchronous, so the reduced timeline cannot change between pages. + expect(events.filter((event) => event.type === 'batch')).toHaveLength(10) + expect(render).toHaveBeenCalledTimes(1) + }) + + it('keeps an exact final page final and preserves unlimited journal readers', async () => { + const journal = await seedJournal(6) + const cursor = { epoch: journal.epoch, sequence: 1 } + expect(journal.readSince(cursor)).toMatchObject({ ok: true, rows: expect.any(Array) }) + const first = readAgentSessionHistory(journal, { + sessionId: identity.sessionId, + direction: 'after', + cursor, + limit: 3 + }) + expect(first).toMatchObject({ ok: true, page: { hasNewer: true } }) + if (!first.ok) { + throw new Error('Expected first page') + } + const last = readAgentSessionHistory(journal, { + sessionId: identity.sessionId, + direction: 'after', + cursor: first.page.window.nextCursor, + limit: 3 + }) + expect(last).toMatchObject({ ok: true, page: { hasNewer: false } }) + const unlimited = journal.readSince(cursor) + expect(unlimited.ok && unlimited.rows).toHaveLength(6) + }) + + it('reports a sequence gap when the next page reaches it', async () => { + const journal = await seedJournal(6) + const { db } = openJournalDatabase(journalDatabaseFile(root!)) + try { + db.prepare('DELETE FROM journal_rows WHERE session_id = ? AND seq = ?').run( + identity.sessionId, + 4 + ) + } finally { + db.close() + } + const first = readAgentSessionHistory(journal, { + sessionId: identity.sessionId, + direction: 'after', + cursor: { epoch: journal.epoch, sequence: 1 }, + limit: 2 + }) + expect(first).toMatchObject({ ok: true, page: { hasNewer: true } }) + if (!first.ok) { + throw new Error('Expected first page') + } + expect( + readAgentSessionHistory(journal, { + sessionId: identity.sessionId, + direction: 'after', + cursor: first.page.window.nextCursor, + limit: 2 + }) + ).toMatchObject({ ok: false, reset: 'journal_gap' }) + }) + + it.each(['{', '{"v":9999}'])( + 'preserves parse-stop behavior at lookahead: %s', + async (rowJson) => { + const journal = await seedJournal(6) + const { db } = openJournalDatabase(journalDatabaseFile(root!)) + try { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE session_id = ? AND seq = ?').run( + rowJson, + identity.sessionId, + 4 + ) + } finally { + db.close() + } + const page = readAgentSessionHistory(journal, { + sessionId: identity.sessionId, + direction: 'after', + cursor: { epoch: journal.epoch, sequence: 1 }, + limit: 2 + }) + expect(page).toMatchObject({ + ok: true, + page: { hasNewer: false, window: { nextCursor: { sequence: 3 } } } + }) + if (!page.ok) { + throw new Error('Expected valid prefix') + } + expect(page.page.items.map((item) => item.itemId)).toEqual(['item-0', 'item-1']) + } + ) +}) diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-page-bounds.ts b/src/main/native-chat/agent-session-wire/agent-session-history-page-bounds.ts index df28b234f7c..f305b9cc149 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-history-page-bounds.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-history-page-bounds.ts @@ -22,15 +22,28 @@ export function historyEntryBytes( return Buffer.byteLength(JSON.stringify(item), 'utf8') + (submissionBytes.get(item.itemId) ?? 0) } +// Keyed on the snapshot's own submissions array, which the reducer rebuilds on +// every change, so a paged read over one snapshot serializes submissions once. +const bytesBySubmissions = new WeakMap< + readonly AgentJournalSubmission[], + ReadonlyMap +>() + export function submissionBytesByItemId( submissions: readonly AgentJournalSubmission[] -): Map { - return new Map( +): ReadonlyMap { + const cached = bytesBySubmissions.get(submissions) + if (cached) { + return cached + } + const bytes = new Map( submissions.map((submission) => [ agentJournalSubmissionKey(submission.clientMessageId), Buffer.byteLength(JSON.stringify(submission), 'utf8') ]) ) + bytesBySubmissions.set(submissions, bytes) + return bytes } export function oversizedHistoryItem( @@ -53,8 +66,7 @@ export function boundHistoryItemsByBytes( submissionBytes: ReadonlyMap, maxBytes: number ): { items: AgentJournalRenderItem[]; dropped: number } { - const groups = groupItemsBySequence(items) - const ordered = keep === 'newest' ? groups.toReversed() : groups + const ordered = groupItemsBySequence(items, keep) const kept: AgentJournalRenderItem[][] = [] let total = 0 for (const group of ordered) { @@ -75,29 +87,42 @@ export function boundHistoryItemsByBytes( } } -function groupItemsBySequence( - items: readonly AgentJournalRenderItem[] -): AgentJournalRenderItem[][] { - const groups: AgentJournalRenderItem[][] = [] - for (const item of items) { - const current = groups.at(-1) - if (current?.[0]?.sequence === item.sequence) { - current.push(item) +/** Adjacent same-sequence runs, walked from the end (`newest`) or the start (`oldest`) + * so a caller that stops at its window never groups the history it will not return. + * Groups and their items keep the order the eager forward grouping produced. */ +function* groupItemsBySequence( + items: readonly AgentJournalRenderItem[], + keep: 'newest' | 'oldest' +): Generator { + let cursor = keep === 'newest' ? items.length : 0 + while (keep === 'newest' ? cursor > 0 : cursor < items.length) { + if (keep === 'newest') { + let start = cursor - 1 + const sequence = items[start].sequence + while (start > 0 && items[start - 1].sequence === sequence) { + start -= 1 + } + yield items.slice(start, cursor) + cursor = start } else { - groups.push([item]) + let end = cursor + 1 + const sequence = items[cursor].sequence + while (end < items.length && items[end].sequence === sequence) { + end += 1 + } + yield items.slice(cursor, end) + cursor = end } } - return groups } export function newestWholeSequenceGroups( items: readonly AgentJournalRenderItem[], limit: number ): AgentJournalRenderItem[] { - const groups = groupItemsBySequence(items) const selected: AgentJournalRenderItem[][] = [] let count = 0 - for (const group of groups.toReversed()) { + for (const group of groupItemsBySequence(items, 'newest')) { if (selected.length > 0 && count + group.length > limit) { break } diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-page-grouping-parity.test.ts b/src/main/native-chat/agent-session-wire/agent-session-history-page-grouping-parity.test.ts new file mode 100644 index 00000000000..e1e8f68f7bd --- /dev/null +++ b/src/main/native-chat/agent-session-wire/agent-session-history-page-grouping-parity.test.ts @@ -0,0 +1,171 @@ +import { expect, it } from 'vitest' +import type { AgentJournalRenderItem } from '../../../shared/agent-session-journal-types' +import { + boundHistoryItemsByBytes, + historyEntryBytes, + newestWholeSequenceGroups, + oversizedHistoryItem +} from './agent-session-history-page-bounds' + +/** The pre-generator eager grouping, verbatim, as the differential oracle. */ +function referenceGroups(items: readonly AgentJournalRenderItem[]): AgentJournalRenderItem[][] { + const groups: AgentJournalRenderItem[][] = [] + for (const item of items) { + const current = groups.at(-1) + if (current?.[0]?.sequence === item.sequence) { + current.push(item) + } else { + groups.push([item]) + } + } + return groups +} + +function referenceNewestWholeSequenceGroups( + items: readonly AgentJournalRenderItem[], + limit: number +): AgentJournalRenderItem[] { + const selected: AgentJournalRenderItem[][] = [] + let count = 0 + for (const group of referenceGroups(items).toReversed()) { + if (selected.length > 0 && count + group.length > limit) { + break + } + selected.push(group) + count += group.length + } + return selected.toReversed().flat() +} + +function referenceBoundHistoryItemsByBytes( + items: AgentJournalRenderItem[], + keep: 'newest' | 'oldest', + submissionBytes: ReadonlyMap, + maxBytes: number +): { items: AgentJournalRenderItem[]; dropped: number } { + const groups = referenceGroups(items) + const ordered = keep === 'newest' ? groups.toReversed() : groups + const kept: AgentJournalRenderItem[][] = [] + let total = 0 + for (const group of ordered) { + const bytes = group.reduce((sum, item) => sum + historyEntryBytes(item, submissionBytes), 0) + if (kept.length === 0 && bytes > maxBytes) { + kept.push(group.map((item) => oversizedHistoryItem(item, bytes))) + break + } + if (total + bytes > maxBytes) { + break + } + kept.push(group) + total += bytes + } + return { + items: (keep === 'newest' ? kept.toReversed() : kept).flat(), + dropped: items.length - kept.reduce((count, group) => count + group.length, 0) + } +} + +function item(index: number, sequence: number): AgentJournalRenderItem { + return { + itemId: `i${index}`, + revision: 1, + sequence, + observedAt: index, + body: { kind: 'status', text: `s${index}` } + } +} + +/** Every sequence-run shape of `length` items, as run-length compositions. */ +function* runShapes(length: number): Generator { + if (length === 0) { + yield [] + return + } + for (let first = 1; first <= length; first += 1) { + for (const rest of runShapes(length - first)) { + yield [first, ...rest] + } + } +} + +/** Build items from run lengths; `repeatSequence` reuses an earlier sequence value + * in a later run so non-adjacent duplicates are exercised too. */ +function buildItems(runs: number[], repeatSequence: boolean): AgentJournalRenderItem[] { + const items: AgentJournalRenderItem[] = [] + let index = 0 + runs.forEach((runLength, runIndex) => { + const sequence = repeatSequence && runIndex > 0 && runIndex % 2 === 0 ? 0 : runIndex + for (let i = 0; i < runLength; i += 1) { + items.push(item(index++, sequence)) + } + }) + return items +} + +it('matches eager grouping at every newest-window limit for every run shape', () => { + let cases = 0 + for (let length = 0; length <= 7; length += 1) { + for (const runs of runShapes(length)) { + for (const repeatSequence of [false, true]) { + const items = buildItems(runs, repeatSequence) + // Every boundary, including 0, each exact group edge, and past the end. + for (let limit = 0; limit <= length + 1; limit += 1) { + expect( + newestWholeSequenceGroups(items, limit), + `runs ${runs.join(',')} repeat ${repeatSequence} limit ${limit}` + ).toEqual(referenceNewestWholeSequenceGroups(items, limit)) + cases += 1 + } + } + } + } + expect(cases).toBeGreaterThan(1000) +}) + +it('matches eager byte bounding at every budget boundary in both directions', () => { + const submissionBytes = new Map() + let truncatedCases = 0 + let partialCases = 0 + for (let length = 1; length <= 6; length += 1) { + for (const runs of runShapes(length)) { + for (const repeatSequence of [false, true]) { + const items = buildItems(runs, repeatSequence) + const perItem = historyEntryBytes(items[0]!, submissionBytes) + // Sweep exact group-boundary budgets plus one byte either side of each. + const budgets = new Set([0, 1]) + for (let n = 0; n <= length + 1; n += 1) { + budgets.add(n * perItem - 1) + budgets.add(n * perItem) + budgets.add(n * perItem + 1) + } + for (const keep of ['newest', 'oldest'] as const) { + for (const maxBytes of budgets) { + const actual = boundHistoryItemsByBytes([...items], keep, submissionBytes, maxBytes) + const expected = referenceBoundHistoryItemsByBytes( + [...items], + keep, + submissionBytes, + maxBytes + ) + expect( + actual, + `runs ${runs.join(',')} repeat ${repeatSequence} keep ${keep} bytes ${maxBytes}` + ).toEqual(expected) + if ( + actual.items.some( + (entry) => entry.body.kind === 'status' && /truncated/.test(entry.body.text) + ) + ) { + truncatedCases += 1 + } else if (actual.dropped > 0) { + partialCases += 1 + } + } + } + } + } + } + // The oversized-first-group and partial-window paths must both be exercised. + expect(truncatedCases).toBeGreaterThan(50) + expect(partialCases).toBeGreaterThan(50) +}) diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-page-scaling.test.ts b/src/main/native-chat/agent-session-wire/agent-session-history-page-scaling.test.ts new file mode 100644 index 00000000000..794966b017f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/agent-session-history-page-scaling.test.ts @@ -0,0 +1,48 @@ +import { expect, it } from 'vitest' +import type { AgentJournalRenderItem } from '../../../shared/agent-session-journal-types' +import { + boundHistoryItemsByBytes, + newestWholeSequenceGroups +} from './agent-session-history-page-bounds' + +function item(index: number): AgentJournalRenderItem { + return { + itemId: String(index), + revision: 1, + sequence: index, + observedAt: index, + body: { kind: 'status', text: 'ok' } + } +} + +it('visits only the retained sequence window and its boundary', () => { + let reads = 0 + const items = Array.from({ length: 10000 }, (_, index) => ({ + ...item(index), + get sequence() { + reads += 1 + return index + } + })) + expect(newestWholeSequenceGroups(items, 100).map((entry) => entry.itemId)).toEqual( + Array.from({ length: 100 }, (_, index) => String(9900 + index)) + ) + expect(reads).toBeLessThan(300) + reads = 0 + expect(boundHistoryItemsByBytes(items, 'newest', new Map(), 1000).items.length).toBeGreaterThan(0) + expect(reads).toBeLessThan(100) +}) + +it('retains entire boundary groups and preserves oversized first-group truncation', () => { + const items = [item(1), { ...item(2), sequence: 1 }, item(3), { ...item(4), sequence: 3 }] + expect(newestWholeSequenceGroups(items, 1)).toEqual(items.slice(2)) + expect(newestWholeSequenceGroups(items, 3)).toEqual(items.slice(2)) + expect(boundHistoryItemsByBytes(items, 'oldest', new Map(), 1)).toMatchObject({ + dropped: 2, + items: [{ itemId: '1' }, { itemId: '2' }] + }) + expect(boundHistoryItemsByBytes(items, 'newest', new Map(), 1)).toMatchObject({ + dropped: 2, + items: [{ itemId: '3' }, { itemId: '4' }] + }) +}) diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-page.ts b/src/main/native-chat/agent-session-wire/agent-session-history-page.ts index 35e18f792a7..231b0b248f7 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-history-page.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-history-page.ts @@ -45,9 +45,11 @@ export function resolveHistoryLimit(limit: number | undefined): number { export function readAgentSessionHistory( journal: AgentSessionJournal, - request: AgentSessionHistoryRequest + request: AgentSessionHistoryRequest, + /** Reduced state to read against. A synchronous multi-page catch-up passes one + * snapshot for the whole run so each page costs its own rows, not the timeline. */ + snapshot: AgentJournalSnapshot = journal.snapshot() ): AgentSessionHistoryResult { - const snapshot = journal.snapshot() if (journal.isReadOnly) { return historyReset(snapshot, 'schema_unreadable') } @@ -90,6 +92,24 @@ export function readAgentSessionHistory( } } +/** + * A catch-up run over one journal. Pages share one reduced timeline, so the run + * costs its own rows instead of re-reducing every item per page; the cursor + * check re-reduces if anything did advance the journal between pages. + */ +export function createAgentSessionCatchUpReader( + journal: AgentSessionJournal +): (request: AgentSessionHistoryRequest) => AgentSessionHistoryResult { + let snapshot = journal.snapshot() + return (request) => { + const live = journal.cursor() + if (live.epoch !== snapshot.cursor.epoch || live.sequence !== snapshot.cursor.sequence) { + snapshot = journal.snapshot() + } + return readAgentSessionHistory(journal, request, snapshot) + } +} + export function readAgentSessionHydrationPage( journal: AgentSessionJournal, fence?: number @@ -145,7 +165,8 @@ function readForward( // a page it cannot place. return historyReset(snapshot, 'cursor_ahead') } - const since = journal.readSince(cursor) + // One lookahead preserves hasNewer without rereading the entire remaining journal per page. + const since = journal.readSince(cursor, limit + 1) if (!since.ok) { return historyReset(snapshot, since.reset) } diff --git a/src/main/native-chat/agent-session-wire/agent-session-journal-batch.ts b/src/main/native-chat/agent-session-wire/agent-session-journal-batch.ts index 9e7b304338e..57f2291e476 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-journal-batch.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-journal-batch.ts @@ -6,7 +6,11 @@ // key instead of appearing as a second copy of the user's own message. import { agentJournalSubmissionKey } from '../../../shared/agent-session-journal-item-key' -import type { AgentJournalSnapshot } from '../../../shared/agent-session-journal-types' +import type { + AgentJournalRenderItem, + AgentJournalSnapshot, + AgentJournalSubmission +} from '../../../shared/agent-session-journal-types' import type { AgentSessionJournalBatch } from '../../../shared/agent-session-wire' import { findSequenceGap } from '../agent-session-journal/journal-cursor' import type { JournalRow } from '../agent-session-journal/journal-row-schema' @@ -31,7 +35,7 @@ export function projectJournalBatch(input: { if (gap) { return { ok: false, reset: 'journal_gap' } } - const aliases = submissionAliases(input.snapshot) + const aliases = submissionAliases(input.snapshot.submissions) const touchedItemIds = new Set() const touchedClientMessageIds = new Set() for (const row of input.rows) { @@ -57,7 +61,7 @@ export function projectJournalBatch(input: { } } - const live = new Map(input.snapshot.items.map((item) => [item.itemId, item])) + const live = liveItemsById(input.snapshot.items) const items = [...touchedItemIds] .map((itemId) => live.get(itemId)) .filter((item) => item !== undefined) @@ -75,18 +79,49 @@ export function projectJournalBatch(input: { } } +// Both indexes are keyed on the snapshot arrays themselves, which the reducer +// rebuilds on every change, so a paged catch-up over one snapshot pays for them +// once instead of once per page — including the byte-shrink loop's re-projections. +const liveItemsByTimeline = new WeakMap< + readonly AgentJournalRenderItem[], + ReadonlyMap +>() +const aliasesBySubmissions = new WeakMap< + readonly AgentJournalSubmission[], + ReadonlyMap +>() + +function liveItemsById( + items: readonly AgentJournalRenderItem[] +): ReadonlyMap { + const cached = liveItemsByTimeline.get(items) + if (cached) { + return cached + } + const live = new Map(items.map((item) => [item.itemId, item])) + liveItemsByTimeline.set(items, live) + return live +} + /** * Provider item id → the submission slot that adopted it, rebuilt from the * snapshot's own accepted submissions. This mirrors the alias the reducer * writes on an accepted dispatch; deriving it here keeps the projection a pure * function of published state instead of reaching into reducer internals. */ -function submissionAliases(snapshot: AgentJournalSnapshot): Map { +function submissionAliases( + submissions: readonly AgentJournalSubmission[] +): ReadonlyMap { + const cached = aliasesBySubmissions.get(submissions) + if (cached) { + return cached + } const aliases = new Map() - for (const submission of snapshot.submissions) { + for (const submission of submissions) { if (submission.dispatchState === 'accepted' && submission.providerItemId) { aliases.set(submission.providerItemId, agentJournalSubmissionKey(submission.clientMessageId)) } } + aliasesBySubmissions.set(submissions, aliases) return aliases } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts index 412aa88025d..986fcefadbf 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts @@ -38,4 +38,5 @@ export type StructuredAgentSessionAttachContext = { reconcileLeases: (sessionId: string) => Promise serialize: (sessionId: string, task: () => Promise) => Promise now: () => number + publishStatus?: (sessionId: string) => void } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-recovery.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-recovery.ts index 5f9894c3280..728942b6347 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-recovery.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-recovery.ts @@ -22,6 +22,7 @@ export class StructuredAgentSessionEventRecovery { sessions: Map flushLifecycle: (sessionId: string) => Promise publishFence: (sessionId: string, session: StructuredAgentSessionHostSession) => void + publishStatus?: (sessionId: string) => void hasResumeCapableHolder: (sessionId: string) => boolean serialize: (sessionId: string, task: () => Promise) => Promise now: () => number diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts index 7c8c65a5292..3495409f127 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts @@ -25,6 +25,7 @@ type HostHandoffAccess = { flush: (sessionId: string) => Promise serialize: (sessionId: string, task: () => Promise) => Promise subscribers: AgentSessionSubscribers + publishStatus?: (sessionId: string) => void now: () => number } @@ -82,6 +83,7 @@ export function createStructuredAgentSessionHostHandoff( return { state: 'live' } } host.session(sessionId).hasProviderChild = false + host.publishStatus?.(sessionId) try { await host.flush(sessionId) host.eventSink(sessionId).unbind() @@ -243,6 +245,7 @@ export async function acquireNativeHandoffOwner( return rethrowAfterAgentSessionAcquisitionCleanup(deps.adapter, input.sessionId, error) } session.hasProviderChild = true + host.publishStatus?.(input.sessionId) session.fence = proved.lease.runtimeFence session.acquisitionGeneration = acquired.acquisitionGeneration ?? null eventSink.bind({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts index ba62db1704e..d1664bc985b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts @@ -93,13 +93,19 @@ export async function resumeStructuredAgentSessionForHold( export function createStructuredAgentSessionHolds( context: StructuredAgentSessionLifetimeContext, input: { - resume: (sessionId: string) => Promise - evict: (sessionId: string) => Promise + reconcileLeases: (sessionId: string) => Promise + attach: Parameters[0]['attach'] + close: (sessionId: string) => Promise } ): StructuredAgentSessionHolds { return new StructuredAgentSessionHolds({ - resume: input.resume, - evict: input.evict, + resume: (sessionId) => + resumeStructuredAgentSessionForHold( + { ...context, reconcileLeases: input.reconcileLeases }, + sessionId, + input.attach + ), + evict: input.close, hasProviderChild: (sessionId) => hasProviderChild(context, sessionId), isTurnActive: (sessionId) => { const session = context.sessions.get(sessionId) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 257c7a4e4f7..2df2fdf5412 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -27,7 +27,6 @@ import { attachStructuredAgentSession } from './structured-agent-session-attach- import { createStructuredAgentSessionHolds, evictHeldStructuredAgentSession, - resumeStructuredAgentSessionForHold, type StructuredAgentSessionLifetimeContext } from './structured-agent-session-host-lifetime' import type { @@ -118,16 +117,13 @@ export class StructuredAgentSessionHost { flush: (sessionId) => this.flushStreamedEvents(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), subscribers: this.subscribers, + publishStatus: (sessionId) => this.statusFeed.publish(sessionId), now: this.now }) this.holds = createStructuredAgentSessionHolds(this.lifetimeContext(), { - resume: (sessionId) => - resumeStructuredAgentSessionForHold( - { ...this.lifetimeContext(), reconcileLeases: this.reconcileLeases }, - sessionId, - (params) => this.attach({ callerKey: 'trusted-local:surface-hold' }, params) - ), - evict: (sessionId) => this.close(sessionId) + reconcileLeases: this.reconcileLeases, + attach: (params) => this.attach({ callerKey: 'trusted-local:surface-hold' }, params), + close: (sessionId) => this.close(sessionId) }) this.restore = createStructuredAgentSessionHostRestore(deps, this.sessions, () => this.now(), { reconcile: this.reconcileLeases, @@ -149,6 +145,7 @@ export class StructuredAgentSessionHost { flushLifecycle: (sessionId) => this.runtimeState.lifecycleBarrier(sessionId), publishFence: (sessionId, session) => this.subscribers.snapshot(sessionId, session.journal, session.fence), + publishStatus: (sessionId) => this.statusFeed.publish(sessionId), hasResumeCapableHolder: (sessionId) => this.holds.hasResumeCapableHolder(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), now: () => this.now(), @@ -193,7 +190,8 @@ export class StructuredAgentSessionHost { subscribers: this.subscribers, tasks: this.tasks, reconcileLeases: (sessionId) => this.reconcileLeases(sessionId), - serialize: (sessionId, task) => this.serialize(sessionId, task) + serialize: (sessionId, task) => this.serialize(sessionId, task), + publishStatus: (sessionId) => this.statusFeed.publish(sessionId) } } /** Releases a session's resources without ending the conversation: the record and journal stay @@ -202,6 +200,7 @@ export class StructuredAgentSessionHost { return this.serialize(sessionId, async () => { await this.handoffs.closeRetainedTuiOwner(sessionId) await evictHeldStructuredAgentSession(this.lifetimeContext(), sessionId) + this.statusFeed.revokeLive(sessionId) // Whoever asked for the close, the surfaces that were holding this session are looking at a // session that no longer exists. A failed eviction throws above and keeps them. this.holds.forget(sessionId) @@ -213,8 +212,11 @@ export class StructuredAgentSessionHost { listSessionTabs = () => listStructuredAgentSessionTabs(this.sessions) - getPersistedVisibleSessionTabIndex = (): { present: boolean; sessionIds: string[] } => - this.deps.store.getVisibleSessionTabIndex() + /** Last projected status for every structured session this host still holds, for non-subscribing + * readers. The retained projections of forgotten sessions are deliberately not included. */ + readonly liveSessionStatusSummaries = () => this.statusFeed.liveSessionSummaries() + + getPersistedVisibleSessionTabIndex = () => this.deps.store.getVisibleSessionTabIndex() setSessionTabVisibility = (sessionId: string, visible: boolean): Promise => this.deps.store.setSessionTabVisibility(sessionId, visible) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts index c1b52f879d4..1e3b9bb25a3 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts @@ -53,15 +53,24 @@ async function openJournal(sessionId = SESSION, now?: () => number) { }) } -function indexed(session: { journal: Awaited> }) { +function indexed(session: { + journal: Awaited> + hasProviderChild?: boolean +}) { return { journal: session.journal, + ...(session.hasProviderChild !== undefined + ? { hasProviderChild: session.hasProviderChild } + : {}), params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' as const } } } function feedFor( - sessions: Map> }>, + sessions: Map< + string, + { journal: Awaited>; hasProviderChild?: boolean } + >, record: Partial | null = null, onStatusChanged?: StructuredAgentSessionStatusFeedDeps['onStatusChanged'] ) { @@ -88,6 +97,41 @@ function feedFor( } describe('StructuredAgentSessionStatusFeed', () => { + it('publishes provider ownership transitions without changing journal time', async () => { + const journal = await openJournal() + const sessions = new Map([[SESSION, { journal, hasProviderChild: true }]]) + const { feed, events, dispose } = feedFor(sessions) + events.length = 0 + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + feed.publish(SESSION, journal) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ hostExecutionOwned: true, updatedAt: expect.any(Number) }) + }) + const firstStatus = events.at(-1) + expect(firstStatus?.type).toBe('status') + if (firstStatus?.type !== 'status') { + throw new Error('status publication missing') + } + const journalTime = firstStatus.session.updatedAt + sessions.get(SESSION)!.hasProviderChild = false + feed.publish(SESSION, journal) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'idle', updatedAt: journalTime }) + }) + const secondStatus = events.at(-1) + expect(secondStatus?.type).toBe('status') + if (secondStatus?.type === 'status') { + expect(secondStatus.session).not.toHaveProperty('hostExecutionOwned') + } + dispose() + }) + it('opens with every readable session and reports no status before a persisted turn', async () => { const journal = await openJournal() const { events } = feedFor(new Map([[SESSION, { journal }]])) @@ -523,3 +567,37 @@ describe('StructuredAgentSessionStatusFeed', () => { }) }) }) + +/** + * `published` is a broadcast cache, not a roster. It deliberately never retracts — an evicted idle + * session is still idle, and a reloading renderer must not lose every settled row — so enumerating + * it lists every session this host has ever opened. Eviction's `forget-session` step deletes the + * session from the live map and touches nothing else, so a poller has to intersect with that map. + */ +describe('the polling reader answers from the live sessions, not the retained cache', () => { + it('drops an evicted session from the poll while a late subscriber still sees it', async () => { + const journal = await openJournal() + const sessions = new Map([[SESSION, { journal }]]) + const { feed } = feedFor(sessions) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + feed.publish(SESSION, journal) + expect(feed.liveSessionSummaries().map((summary) => summary.sessionId)).toEqual([SESSION]) + + // Exactly what eviction's `forget-session` step does; nothing else touches the feed. + sessions.delete(SESSION) + + expect(feed.liveSessionSummaries()).toEqual([]) + const late: AgentSessionStatusEvent[] = [] + feed.subscribe({ id: 'list-2', emit: (event) => late.push(event) }) + expect(late).toEqual([ + { + type: 'snapshot', + sessions: [expect.objectContaining({ sessionId: SESSION, status: 'idle' })] + } + ]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts index cfd85ff4648..61d9649587e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts @@ -30,6 +30,7 @@ export type StructuredAgentSessionStatusSubscriber = { type StatusFeedSession = { journal: AgentSessionJournal params: { location: { workspaceId: string }; provider: AgentSessionRecord['provider'] } + hasProviderChild?: boolean } export type StructuredAgentSessionStatusFeedDeps = { @@ -46,6 +47,7 @@ function summariesEqual(a: AgentSessionStatusSummary, b: AgentSessionStatusSumma a.workspaceId === b.workspaceId && a.agent === b.agent && a.status === b.status && + a.hostExecutionOwned === b.hostExecutionOwned && a.rewindBlockedReason === b.rewindBlockedReason && // Settled activity changes ranking; streaming active turns must stay quiet. (a.status !== 'idle' || a.updatedAt === b.updatedAt) && @@ -76,6 +78,27 @@ export class StructuredAgentSessionStatusFeed { return () => this.unsubscribe(subscriber.id) } + /** + * Summaries for the sessions this host still holds, for readers that poll instead of subscribing. + * + * `published` never retracts, so it is a broadcast cache and not a roster: enumerating it lists + * every session ever opened here. A caller asking what is running gets the live intersection, + * while the retained view a subscriber opens on stays whole. + * + * Deliberately does NOT re-project: a subscriber's snapshot is the live read, and re-running the + * journal reduction per caller would make an enumerating command pay for every session it lists. + */ + liveSessionSummaries(): AgentSessionStatusSummary[] { + const summaries: AgentSessionStatusSummary[] = [] + for (const [sessionId] of this.deps.sessions) { + const summary = this.published.get(sessionId) + if (summary) { + summaries.push(summary) + } + } + return summaries + } + unsubscribe(id: string): void { const subscriber = this.subscribers.get(id) if (!subscriber) { @@ -89,6 +112,20 @@ export class StructuredAgentSessionStatusFeed { } } + /** Revoke live execution authority while retaining the last projection for reload history. */ + revokeLive(sessionId: string): void { + const previous = this.published.get(sessionId) + if (!previous) { + return + } + const { hostExecutionOwned: _hostExecutionOwned, ...retained } = previous + this.published.set(sessionId, retained) + this.broadcast({ + type: 'status', + session: retained + }) + } + /** Re-projects one session after its journal changed; equal projections are not re-sent. */ publish(sessionId: string, journal?: AgentSessionJournal, options?: { replay?: boolean }): void { const session = this.deps.sessions.get(sessionId) @@ -126,6 +163,7 @@ export class StructuredAgentSessionStatusFeed { sessionId, workspaceId: session.params.location.workspaceId, agent: session.params.provider, + ...(session.hasProviderChild ? { hostExecutionOwned: true as const } : {}), ...projectStructuredAgentSessionStatusSummary(items), ...(record?.rewind?.phase === 'prepared' || record?.rewind?.phase === 'provider-succeeded' ? { rewindBlockedReason: 'outcome-unknown' as const } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts index ba439e6427b..a5062373dad 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts @@ -18,7 +18,7 @@ import { } from '../../../shared/agent-session-wire' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import { - readAgentSessionHistory, + createAgentSessionCatchUpReader, readAgentSessionHydrationPage } from './agent-session-history-page' @@ -229,14 +229,13 @@ export class AgentSessionSubscribers { backgroundTasks?: AgentSessionBackgroundTaskState | null, activity?: AgentSessionTurnActivity | null ): void { - const publishedActivity = - activity !== undefined - ? activity - : emitCheckpoint - ? (this.activityBySession.get(subscriber.sessionId) ?? null) - : undefined + const checkpointActivity = emitCheckpoint + ? this.activityField(subscriber.sessionId).activity + : undefined + const publishedActivity = activity !== undefined ? activity : checkpointActivity + const readPage = createAgentSessionCatchUpReader(journal) while (true) { - const result = readAgentSessionHistory(journal, { + const result = readPage({ sessionId: subscriber.sessionId, direction: 'after', cursor: subscriber.cursor, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts index 87fe0cbbeec..ddc4af9f2d3 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts @@ -32,6 +32,7 @@ export type StructuredAgentSessionUnexpectedExitContext = { sessions: Map flushLifecycle: (sessionId: string) => Promise publishFence: (sessionId: string, session: StructuredAgentSessionHostSession) => void + publishStatus?: (sessionId: string) => void hasResumeCapableHolder: (sessionId: string) => boolean serialize: (sessionId: string, task: () => Promise) => Promise now: () => number @@ -59,6 +60,7 @@ export async function settleUnexpectedStructuredAgentSessionExit( if (!record || record.lease.handoffStage !== null) { // The handoff coordinator owns an already-started transition. session.hasProviderChild = false + context.publishStatus?.(unexpectedEvent.sessionId) return null } @@ -117,6 +119,7 @@ export async function settleUnexpectedStructuredAgentSessionExit( context.onBarrierError?.(unexpectedEvent.sessionId, error) } finally { session.hasProviderChild = false + context.publishStatus?.(unexpectedEvent.sessionId) if (released) { session.fence = released.lease.runtimeFence context.publishFence(unexpectedEvent.sessionId, session) diff --git a/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts b/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts new file mode 100644 index 00000000000..42d03f46b5b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts @@ -0,0 +1,48 @@ +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionBackgroundTaskState } from '../../../shared/agent-session-wire' +import { conversationCommandBlocked } from './structured-conversation-command-admission' +import type { AgentSessionTurnContext } from './structured-agent-session-turns' + +function contextWith( + backgroundTasks: AgentSessionBackgroundTaskState | null +): AgentSessionTurnContext { + return { + sessionId: 'session-1', + journal: { + snapshot: () => ({ items: [] }), + submissions: () => [] + }, + adapter: { backgroundTaskState: () => backgroundTasks } + } as unknown as AgentSessionTurnContext +} + +const RECORD = { lease: {} } as unknown as AgentSessionRecord + +describe('conversationCommandBlocked background tasks', () => { + it('admits the command when nothing is being monitored', () => { + expect(conversationCommandBlocked(contextWith(null), RECORD)).toBeNull() + }) + + it('asks for a stop when the host accepts targeted stops', () => { + const blocked = conversationCommandBlocked( + contextWith({ state: 'monitoring', supportsTaskStop: true }), + RECORD + ) + expect(blocked).toBe('Stop background tasks before using this command.') + }) + + it('asks for a stop on a host that predates the stop-capability field', () => { + const blocked = conversationCommandBlocked(contextWith({ state: 'monitoring' }), RECORD) + expect(blocked).toBe('Stop background tasks before using this command.') + }) + + it('asks the user to wait when the provider exposes no stop at all', () => { + // Codex: an instruction to stop would name a control that does not exist. + const blocked = conversationCommandBlocked( + contextWith({ state: 'monitoring', supportsStopAll: false }), + RECORD + ) + expect(blocked).toBe('Wait for background tasks to finish before using this command.') + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.ts b/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.ts index a69a5163b67..78b9f4a59d0 100644 --- a/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.ts +++ b/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.ts @@ -38,8 +38,14 @@ export function conversationCommandBlocked( ) { return 'Resolve the pending question or approval before using this command.' } - if (ctx.adapter.backgroundTaskState?.(ctx.sessionId)?.state === 'monitoring') { - return 'Stop background tasks before using this command.' + const backgroundTasks = ctx.adapter.backgroundTaskState?.(ctx.sessionId) + if (backgroundTasks?.state === 'monitoring') { + // Only ask for a stop the host can actually perform. A provider that + // exposes neither a targeted nor an untargeted stop would otherwise leave + // the command refused behind an instruction nobody can follow. + return backgroundTasks.supportsTaskStop || backgroundTasks.supportsStopAll !== false + ? 'Stop background tasks before using this command.' + : 'Wait for background tasks to finish before using this command.' } if ( ctx.journal diff --git a/src/main/native-chat/agent-session-wire/structured-rewind-claude-owner.ts b/src/main/native-chat/agent-session-wire/structured-rewind-claude-owner.ts index f1108df181c..13d2b51f284 100644 --- a/src/main/native-chat/agent-session-wire/structured-rewind-claude-owner.ts +++ b/src/main/native-chat/agent-session-wire/structured-rewind-claude-owner.ts @@ -21,6 +21,7 @@ export async function replaceClaudeRewindOwner( return rewindRefusal('outcome-unknown') } session.hasProviderChild = false + context.publishStatus?.(sessionId) const head = agentSessionProviderHandleChainHead( context.deps.store.getRecord(sessionId)!.providerHandleChain )?.handle diff --git a/src/main/native-chat/subagent-entry-id-bounds.test.ts b/src/main/native-chat/subagent-entry-id-bounds.test.ts new file mode 100644 index 00000000000..56d0edece66 --- /dev/null +++ b/src/main/native-chat/subagent-entry-id-bounds.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest' +import { boundSubagentEntryId, MAX_SUBAGENT_ENTRY_ID_CHARS } from './subagent-entry-id-bounds' + +const SHARED_HEAD = 'a'.repeat(MAX_SUBAGENT_ENTRY_ID_CHARS) + +describe('boundSubagentEntryId', () => { + it('leaves an id that already fits untouched', () => { + const id = 'task-1' + expect(boundSubagentEntryId(id)).toBe(id) + expect(boundSubagentEntryId(SHARED_HEAD)).toBe(SHARED_HEAD) + }) + + it('keeps two ids sharing the whole cap-length head distinct', () => { + const first = boundSubagentEntryId(`${SHARED_HEAD}-one`) + const second = boundSubagentEntryId(`${SHARED_HEAD}-two`) + expect(first).not.toBe(second) + expect(first).toHaveLength(MAX_SUBAGENT_ENTRY_ID_CHARS) + expect(second).toHaveLength(MAX_SUBAGENT_ENTRY_ID_CHARS) + }) + + it('is deterministic and a no-op on an already bounded id', () => { + const bounded = boundSubagentEntryId(`${SHARED_HEAD}-one`) + expect(boundSubagentEntryId(`${SHARED_HEAD}-one`)).toBe(bounded) + expect(boundSubagentEntryId(bounded)).toBe(bounded) + }) +}) diff --git a/src/main/native-chat/subagent-entry-id-bounds.ts b/src/main/native-chat/subagent-entry-id-bounds.ts new file mode 100644 index 00000000000..3abf19a4ff6 --- /dev/null +++ b/src/main/native-chat/subagent-entry-id-bounds.ts @@ -0,0 +1,27 @@ +// Bounding a subagent entry's id — a KEY, not display text. +// +// `NativeChatSubagentEntry.id` keys the roster, so clipping it to a fixed prefix +// merges two distinct children whose ids agree that far and associates one's +// state with the other. An id long enough to need bounding only reaches us from +// an imported legacy transcript, but a bound is still required, so keep a head +// for readability plus a digest of the WHOLE id to keep distinct ids distinct. + +import { createHash } from 'node:crypto' + +/** Shared by every site that puts a roster entry on a wire, so a journal-bound + * id survives the RPC and transcript bounds untouched instead of being clipped + * a second time into a different string. */ +export const MAX_SUBAGENT_ENTRY_ID_CHARS = 512 + +const DIGEST_CHARS = 16 + +/** Returns `id` unchanged when it fits, else a head plus a digest suffix whose + * total length is exactly the cap — so bounding a bounded id is a no-op. */ +export function boundSubagentEntryId(id: string): string { + if (id.length <= MAX_SUBAGENT_ENTRY_ID_CHARS) { + return id + } + const digest = createHash('sha256').update(id, 'utf8').digest('base64url').slice(0, DIGEST_CHARS) + const suffix = `…#${digest}` + return `${id.slice(0, MAX_SUBAGENT_ENTRY_ID_CHARS - suffix.length)}${suffix}` +} diff --git a/src/main/observability/local-file-sink-memory.test.ts b/src/main/observability/local-file-sink-memory.test.ts new file mode 100644 index 00000000000..fef5b8d00dd --- /dev/null +++ b/src/main/observability/local-file-sink-memory.test.ts @@ -0,0 +1,132 @@ +import { mkdtempSync, readFileSync, rmSync, statSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + createLocalFileSink, + DROPPED_RECORD_TYPE, + type LocalFileSink +} from './local-file-sink' + +function parseLine(raw: string): Record { + return JSON.parse(raw) as Record +} + +let directory: string +let sink: LocalFileSink | undefined +beforeEach(() => { + directory = mkdtempSync(join(tmpdir(), 'orca-trace-memory-')) + vi.useFakeTimers() +}) +afterEach(() => { + sink?.close() + sink = undefined + vi.useRealTimers() + rmSync(directory, { recursive: true, force: true }) +}) + +function retainedHeap(): number { + if (!globalThis.gc) { + throw new Error('Memory regression requires --expose-gc') + } + globalThis.gc() + globalThis.gc() + return process.memoryUsage().heapUsed +} + +describe('trace sink rejected record retention', () => { + it('keeps small-record byte scans deferred until the batch flush', () => { + const filePath = join(directory, 'trace.ndjson') + sink = createLocalFileSink({ filePath }) + const byteLength = vi.spyOn(Buffer, 'byteLength') + let beforeFlush: number + let afterFlush: number + try { + for (let index = 0; index < 20; index++) { + sink.push({ index, text: '💡漢字' }) + } + beforeFlush = byteLength.mock.calls.length + sink.flush() + afterFlush = byteLength.mock.calls.length + } finally { + byteLength.mockRestore() + } + expect(beforeFlush).toBe(0) + expect(afterFlush).toBe(20) + }) + + it('releases oversized serialized records before the pending batch flushes', () => { + const filePath = join(directory, 'trace.ndjson') + sink = createLocalFileSink({ filePath, maxBytes: 64 * 1024, batchWindowMs: 200 }) + const before = retainedHeap() + for (let index = 0; index < 24; index++) { + sink.push({ index, payload: 'x'.repeat(1024 * 1024) }) + } + const retained = retainedHeap() - before + expect(statSync(filePath).size).toBe(0) + expect(vi.getTimerCount()).toBe(1) + expect(retained).toBeLessThan(5 * 1024 * 1024) + sink.push({ valid: true }) + vi.advanceTimersByTime(200) + const written = readFileSync(filePath, 'utf8').split('\n').filter(Boolean).map(parseLine) + // Each rejected record leaves a marker, so the gap is readable instead of silent. + expect(written.filter((entry) => entry.type === DROPPED_RECORD_TYPE)).toHaveLength(24) + expect(written.at(-1)).toEqual({ valid: true }) + }) + + it('names the dropped record in the marker instead of leaving a silent gap', () => { + const filePath = join(directory, 'trace.ndjson') + sink = createLocalFileSink({ filePath, maxBytes: 64 * 1024, flushBufferThreshold: 1 }) + sink.push({ + type: 'effect-span', + name: 'worktree.create', + traceId: 'a'.repeat(32), + payload: 'x'.repeat(1024 * 1024) + }) + const [marker] = readFileSync(filePath, 'utf8').split('\n').filter(Boolean).map(parseLine) + expect(marker).toMatchObject({ + type: DROPPED_RECORD_TYPE, + reason: 'oversize', + name: 'worktree.create', + traceId: 'a'.repeat(32) + }) + expect(marker.droppedChars).toBeGreaterThan(1024 * 1024) + // Timestamped so the bundle collector's lookback filter ages markers out like any other span. + expect(BigInt(marker.endTimeUnixNano as string)).toBeGreaterThan(0n) + }) + + it('omits the marker when even the marker would exceed the byte cap', () => { + const filePath = join(directory, 'trace.ndjson') + sink = createLocalFileSink({ filePath, maxBytes: 40, flushBufferThreshold: 1 }) + sink.push({ payload: 'x'.repeat(1_000) }) + sink.push({ ok: 1 }) + expect(readFileSync(filePath, 'utf8')).toBe('{"ok":1}\n') + }) + + it('keeps the same count-triggered flush for valid records beside rejected ones', () => { + const filePath = join(directory, 'trace.ndjson') + sink = createLocalFileSink({ filePath, maxBytes: 32, flushBufferThreshold: 3 }) + sink.push({ valid: 1 }) + sink.push({ payload: '💡'.repeat(20) }) + expect(statSync(filePath).size).toBe(0) + sink.push({ payload: 'x'.repeat(100) }) + expect(readFileSync(filePath, 'utf8')).toBe('{"valid":1}\n') + sink.push({ valid: 2 }) + vi.advanceTimersByTime(200) + expect(readFileSync(filePath, 'utf8')).toBe('{"valid":1}\n{"valid":2}\n') + }) + + it('accepts the exact UTF-8 byte cap and preserves rotation and close flushes', () => { + const filePath = join(directory, 'trace.ndjson') + const record = { text: '💡' } + const line = `${JSON.stringify(record)}\n` + sink = createLocalFileSink({ filePath, maxBytes: Buffer.byteLength(line), maxFiles: 2 }) + sink.push(record) + sink.push({ text: '💡x' }) + sink.push(record) + sink.close() + expect(readFileSync(filePath, 'utf8')).toBe(line) + expect(readFileSync(`${filePath}.1`, 'utf8')).toBe(line) + expect(vi.getTimerCount()).toBe(0) + }) +}) diff --git a/src/main/observability/local-file-sink.ts b/src/main/observability/local-file-sink.ts index b2a74bbbb40..7e6eef9ddac 100644 --- a/src/main/observability/local-file-sink.ts +++ b/src/main/observability/local-file-sink.ts @@ -25,6 +25,9 @@ export const DEFAULT_MAX_FILES = 10 export const DEFAULT_BATCH_WINDOW_MS = 200 const PRIVATE_DIRECTORY_MODE = 0o700 const PRIVATE_FILE_MODE = 0o600 +/** NDJSON `type` for the placeholder left behind when a record is too large to store. */ +export const DROPPED_RECORD_TYPE = 'trace-record-dropped' +const MAX_MARKER_NAME_CHARS = 120 export type LocalFileSinkOptions = { readonly filePath: string @@ -77,7 +80,7 @@ export function createLocalFileSink(opts: LocalFileSinkOptions): LocalFileSink { let fd: number = openAppend(filePath) let currentBytes: number = safeFstatSize(fd) - let buffer: string[] = [] + let buffer: (string | null)[] = [] let timer: NodeJS.Timeout | null = null let closed = false @@ -171,11 +174,10 @@ export function createLocalFileSink(opts: LocalFileSinkOptions): LocalFileSink { } for (const line of lines) { - const lineBytes = Buffer.byteLength(line, 'utf8') - if (lineBytes > maxBytes) { - // Oversized single span would blow the maxFiles × maxBytes envelope; drop just this record. + if (line === null) { continue } + const lineBytes = Buffer.byteLength(line, 'utf8') if (pendingChunkBytes > 0 && currentBytes + pendingChunkBytes + lineBytes > maxBytes) { flushPendingChunk() } @@ -189,6 +191,27 @@ export function createLocalFileSink(opts: LocalFileSinkOptions): LocalFileSink { flushPendingChunk() } + /** + * Stand-in for a record too large to store. `droppedChars` is UTF-16 units, not bytes: measuring + * bytes is the scan this path exists to skip. Returns null when even the marker exceeds maxBytes. + */ + function oversizeMarker(record: unknown, droppedChars: number): string | null { + const span = + typeof record === 'object' && record !== null ? (record as Record) : {} + const name = typeof span.name === 'string' ? span.name.slice(0, MAX_MARKER_NAME_CHARS) : null + const traceId = typeof span.traceId === 'string' ? span.traceId.slice(0, 32) : null + const marker = `${JSON.stringify({ + type: DROPPED_RECORD_TYPE, + reason: 'oversize', + droppedChars, + // Lets the bundle collector's lookback filter age these out like any other span. + endTimeUnixNano: `${Date.now()}000000`, + ...(name === null ? {} : { name }), + ...(traceId === null ? {} : { traceId }) + })}\n` + return Buffer.byteLength(marker, 'utf8') > maxBytes ? null : marker + } + function ensureTimer(): void { if (timer || closed) { return @@ -216,7 +239,13 @@ export function createLocalFileSink(opts: LocalFileSinkOptions): LocalFileSink { // Redactor handles cycles upstream; a throw here means pre-redact data slipped in — drop rather than crash (best-effort). return } - buffer.push(line) + // UTF-8 uses at most three bytes per UTF-16 unit; small records need no admission scan. + const oversized = + line.length > maxBytes || + (line.length * 3 > maxBytes && Buffer.byteLength(line, 'utf8') > maxBytes) + // Rejected records still occupy a buffer slot (preserving flush timing) but carry a marker + // instead of their payload, so the gap they leave is readable rather than silent. + buffer.push(oversized ? oversizeMarker(record, line.length) : line) if (buffer.length >= flushThreshold) { flushBuffer() } else { diff --git a/src/main/opencode-usage/snapshot-rollups.ts b/src/main/opencode-usage/snapshot-rollups.ts index c60f251914d..84e91d7ae78 100644 --- a/src/main/opencode-usage/snapshot-rollups.ts +++ b/src/main/opencode-usage/snapshot-rollups.ts @@ -1,3 +1,4 @@ +import { highestUsageKey } from '../usage/highest-usage-key' import type { OpenCodeUsageBreakdownKind, OpenCodeUsageBreakdownRow, @@ -47,9 +48,8 @@ export function buildOpenCodeUsageSummary( byProject.set(row.projectLabel, (byProject.get(row.projectLabel) ?? 0) + row.totalTokens) } - const topModel = [...byModel.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null - const topProject = - [...byProject.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null + const topModel = highestUsageKey(byModel) + const topProject = highestUsageKey(byProject) return { scope, diff --git a/src/main/persistence/applying-settings/settings-update-terminal-contrast.test.ts b/src/main/persistence/applying-settings/settings-update-terminal-contrast.test.ts new file mode 100644 index 00000000000..7de32d7b36d --- /dev/null +++ b/src/main/persistence/applying-settings/settings-update-terminal-contrast.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it, vi } from 'vitest' +import type { PersistedState } from '../../../shared/persisted-state-types' +import { updateSettings, type SettingsMutationOperations } from './settings-update' + +function makeOperations(): SettingsMutationOperations { + return { + // Only the fields updateSettings reads; the rest of GlobalSettings is irrelevant to the clamp. + state: { settings: { terminalFontSize: 14 }, repos: [] } as unknown as PersistedState, + bumpLocalWorktreeScanGeneration: vi.fn(), + removeRetainedBlob: vi.fn(), + scheduleSave: vi.fn(), + notifySettingsChanged: vi.fn() + } +} + +// #10754: desktop IPC, the web RPC and the CLI all reach the store through this boundary, and xterm +// throws on a non-finite minimumContrastRatio, so the clamp cannot live in the settings UI alone. +describe('updateSettings terminalMinimumContrastRatio', () => { + it('persists an in-range floor unchanged', () => { + const operations = makeOperations() + + expect( + updateSettings(operations, { terminalMinimumContrastRatio: 1 }).terminalMinimumContrastRatio + ).toBe(1) + expect( + updateSettings(operations, { terminalMinimumContrastRatio: 4.5 }).terminalMinimumContrastRatio + ).toBe(4.5) + }) + + it('clamps a hand-edited value into xterm range', () => { + const operations = makeOperations() + + expect( + updateSettings(operations, { terminalMinimumContrastRatio: 0 }).terminalMinimumContrastRatio + ).toBe(1) + expect( + updateSettings(operations, { terminalMinimumContrastRatio: 500 }).terminalMinimumContrastRatio + ).toBe(21) + }) + + it('drops an unusable value back to automatic rather than storing it', () => { + const operations = makeOperations() + + expect( + updateSettings(operations, { + terminalMinimumContrastRatio: Number.NaN + }).terminalMinimumContrastRatio + ).toBeUndefined() + expect( + updateSettings(operations, { + terminalMinimumContrastRatio: 'off' as unknown as number + }).terminalMinimumContrastRatio + ).toBeUndefined() + }) + + it('clears the override so the automatic floor comes back', () => { + const operations = makeOperations() + + updateSettings(operations, { terminalMinimumContrastRatio: 1 }) + expect( + updateSettings(operations, { terminalMinimumContrastRatio: undefined }) + .terminalMinimumContrastRatio + ).toBeUndefined() + }) + + it('leaves a stored floor alone when an unrelated setting is written', () => { + const operations = makeOperations() + + updateSettings(operations, { terminalMinimumContrastRatio: 1 }) + expect(updateSettings(operations, { terminalFontSize: 15 }).terminalMinimumContrastRatio).toBe( + 1 + ) + }) +}) diff --git a/src/main/persistence/applying-settings/settings-update.ts b/src/main/persistence/applying-settings/settings-update.ts index 20614061bdd..e8a080763da 100644 --- a/src/main/persistence/applying-settings/settings-update.ts +++ b/src/main/persistence/applying-settings/settings-update.ts @@ -9,6 +9,7 @@ import { normalizeTerminalQuickCommands } from '../../../shared/terminal-quick-c import { normalizeTerminalCustomThemes } from '../../../shared/terminal-custom-themes' import { normalizeTerminalCursorStyleDefault } from '../../../shared/terminal-cursor-style-settings' import { normalizeDesktopTerminalScrollbackRows } from '../../../shared/terminal-scrollback-policy' +import { normalizeTerminalMinimumContrastRatio } from '../../../shared/terminal-minimum-contrast-settings' import { normalizeTaskProviderSettings } from '../../../shared/task-providers' import { normalizeOpenInApplications } from '../../../shared/open-in-applications' import { normalizeTerminalShortcutPolicy } from '../../../shared/keybindings' @@ -123,6 +124,13 @@ export function updateSettings( updates.terminalScrollbackRows ) } + // Why here: every writer (desktop IPC, web RPC, CLI) crosses this boundary, so xterm can never be + // handed an out-of-range floor, and undefined stays undefined to mean "automatic" (#10754). + if ('terminalMinimumContrastRatio' in updates) { + sanitizedUpdates.terminalMinimumContrastRatio = normalizeTerminalMinimumContrastRatio( + updates.terminalMinimumContrastRatio + ) + } if ( 'terminalTuiScrollSensitivity' in updates || 'terminalTuiScrollSensitivityDefaultedToOne' in updates diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.test.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.test.ts new file mode 100644 index 00000000000..03badd28f94 --- /dev/null +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.test.ts @@ -0,0 +1,136 @@ +import { describe, expect, it, vi } from 'vitest' +import { getDefaultPersistedState, getDefaultWorkspaceSession } from '../../../shared/constants' +import type { SshRemotePtyLease } from '../../../shared/ssh-types' +import { toComparableRelaySshPtyId } from '../../../shared/ssh-pty-id' +import { clearSshRemotePtyBindingsForLeases } from './ssh-pty-binding-cleanup' + +function fixture(count: number) { + const state = getDefaultPersistedState('/home/test') + state.workspaceSession = getDefaultWorkspaceSession() + state.workspaceSession.tabsByWorktree.wt = Array.from({ length: count }, (_, i) => ({ + id: `tab-${i}`, + worktreeId: 'wt', + ptyId: `pty-${i}`, + title: '', + customTitle: null, + color: null, + sortOrder: i, + createdAt: 1 + })) + const leases: SshRemotePtyLease[] = Array.from({ length: count }, (_, i) => ({ + targetId: 'ssh-one', + ptyId: `pty-${i}`, + tabId: `tab-${i}`, + worktreeId: 'wt', + state: 'detached', + createdAt: 1, + updatedAt: 1 + })) + return { + state, + leases, + toComparablePtyId: vi.fn((_target: string, ptyId: string) => ptyId), + scheduleSave: vi.fn() + } +} + +describe('SSH binding cleanup indexing', () => { + it('normalizes each binding once across a large lease inventory', () => { + const operations = fixture(1000) + expect(clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases)).toBe(true) + expect(operations.toComparablePtyId).toHaveBeenCalledTimes(1000) + expect( + operations.state.workspaceSession!.tabsByWorktree.wt.every((tab) => tab.ptyId === null) + ).toBe(true) + expect(operations.scheduleSave).toHaveBeenCalledTimes(1) + }) + + it('retains foreign hosts and conflicting tab/workspace leases', () => { + const operations = fixture(4) + operations.leases[0].targetId = 'ssh-two' + operations.leases[1].tabId = 'other-tab' + operations.leases[2].worktreeId = 'other-workspace' + delete operations.leases[3].tabId + clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases) + expect(operations.state.workspaceSession!.tabsByWorktree.wt.map((tab) => tab.ptyId)).toEqual([ + 'pty-0', + 'pty-1', + 'pty-2', + null + ]) + }) + + it('matches layout leaves against every lease for a PTY while preserving leaf conflicts', () => { + const operations = fixture(1) + const session = operations.state.workspaceSession! + session.tabsByWorktree.wt[0].ptyId = null + session.terminalLayoutsByTabId['tab-0'] = { + root: null, + activeLeafId: null, + expandedLeafId: null, + ptyIdsByLeafId: { + matched: 'pty-0', + protected: 'pty-0', + wildcard: 'pty-1' + } + } + const lease = operations.leases[0] + operations.leases = [ + { ...lease, leafId: 'wrong' }, + { ...lease, leafId: 'matched' }, + { ...lease, ptyId: 'pty-1' } + ] + expect(clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases)).toBe(true) + expect(session.terminalLayoutsByTabId['tab-0'].ptyIdsByLeafId).toEqual({ + protected: 'pty-0' + }) + expect(operations.toComparablePtyId).toHaveBeenCalledTimes(3) + }) + + it('normalizes app-form binding ids onto the relay-form lease key', () => { + // Leases store the relay-local id; sessions may hold the app-wide "ssh:@@" form. + // The index key is the normalized form, so both spellings still name the same PTY. + const operations = fixture(1) + operations.toComparablePtyId = vi.fn(toComparableRelaySshPtyId) + operations.state.workspaceSession!.tabsByWorktree.wt[0].ptyId = 'ssh:ssh-one@@pty-0' + + expect(clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases)).toBe(true) + expect(operations.state.workspaceSession!.tabsByWorktree.wt[0].ptyId).toBeNull() + }) + + it('keeps a binding whose app-form id names a different SSH target', () => { + // Relay-local ids collide across targets ("pty-0" exists on every host). Clearing ssh-one must + // never scrub a pane still bound to a live ssh-two shell. + const operations = fixture(1) + operations.toComparablePtyId = vi.fn(toComparableRelaySshPtyId) + operations.state.workspaceSession!.tabsByWorktree.wt[0].ptyId = 'ssh:ssh-two@@pty-0' + + expect(clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases)).toBe(false) + expect(operations.state.workspaceSession!.tabsByWorktree.wt[0].ptyId).toBe('ssh:ssh-two@@pty-0') + expect(operations.scheduleSave).not.toHaveBeenCalled() + }) + + it('keeps every binding when no lease names its PTY', () => { + // A bucket miss must fail closed: leak a stale id rather than unbind a live pane. + const operations = fixture(2) + for (const lease of operations.leases) { + lease.ptyId = `unrelated-${lease.ptyId}` + } + + expect(clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases)).toBe(false) + expect(operations.state.workspaceSession!.tabsByWorktree.wt.map((tab) => tab.ptyId)).toEqual([ + 'pty-0', + 'pty-1' + ]) + expect(operations.scheduleSave).not.toHaveBeenCalled() + }) + + it('does not index leases when the session holds no bindings to check', () => { + const operations = fixture(500) + operations.state.workspaceSession!.tabsByWorktree = {} + operations.state.workspaceSession!.terminalLayoutsByTabId = {} + + expect(clearSshRemotePtyBindingsForLeases(operations, 'ssh-one', operations.leases)).toBe(false) + expect(operations.toComparablePtyId).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.ts index 65b879795dc..c467e0fc3ab 100644 --- a/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.ts +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-binding-cleanup.ts @@ -9,8 +9,8 @@ export type SshPtyBindingCleanupOperations = { scheduleSave: () => void } +/** `binding.ptyId` must already be in lease-comparable (relay) form; callers normalize it. */ function sshRemotePtyLeaseMayReferenceBinding( - operations: SshPtyBindingCleanupOperations, lease: SshRemotePtyLease, binding: { ptyId: string @@ -20,8 +20,7 @@ function sshRemotePtyLeaseMayReferenceBinding( leafId?: string } ): boolean { - const bindingPtyId = operations.toComparablePtyId(binding.targetId, binding.ptyId) - if (lease.targetId !== binding.targetId || lease.ptyId !== bindingPtyId) { + if (lease.targetId !== binding.targetId || lease.ptyId !== binding.ptyId) { return false } // Why: target removal is destructive; scrub matching bindings before deleting the lease, else removing the tombstone can revive stale PTY ids. @@ -50,6 +49,33 @@ export function clearSshRemotePtyBindingsForLeases( if (!leases?.length) { return false } + // Keyed by the stored (relay) pty id, which is the only form a lease holds; every lookup below + // normalizes the binding id to that form first, so a bucket miss means "no lease names this pty" + // and the binding is KEPT. Failing closed here leaves a stale id to be retired on reattach, + // where clearing on a bad match would strand a live remote shell behind a respawned pane. + let leasesByPtyId: Map | undefined + const referencesBinding = ( + binding: Parameters[1] + ): boolean => { + if (!leasesByPtyId) { + leasesByPtyId = new Map() + for (const lease of leases) { + if (lease.targetId !== targetId) { + continue + } + const entries = leasesByPtyId.get(lease.ptyId) + if (entries) { + entries.push(lease) + } else { + leasesByPtyId.set(lease.ptyId, [lease]) + } + } + } + const ptyId = operations.toComparablePtyId(binding.targetId, binding.ptyId) + return (leasesByPtyId.get(ptyId) ?? []).some((lease) => + sshRemotePtyLeaseMayReferenceBinding(lease, { ...binding, ptyId }) + ) + } let changed = false const sessions = new Set( [ @@ -62,14 +88,7 @@ export function clearSshRemotePtyBindingsForLeases( for (const tab of tabs) { if ( tab.ptyId && - leases.some((lease) => - sshRemotePtyLeaseMayReferenceBinding(operations, lease, { - ptyId: tab.ptyId!, - worktreeId, - targetId, - tabId: tab.id - }) - ) + referencesBinding({ ptyId: tab.ptyId, worktreeId, targetId, tabId: tab.id }) ) { tab.ptyId = null changed = true @@ -92,16 +111,7 @@ export function clearSshRemotePtyBindingsForLeases( const worktreeId = worktreeIdByTabId.get(tabId) const nextBindings = Object.fromEntries( Object.entries(bindings).filter( - ([leafId, ptyId]) => - !leases.some((lease) => - sshRemotePtyLeaseMayReferenceBinding(operations, lease, { - ptyId, - targetId, - worktreeId, - tabId, - leafId - }) - ) + ([leafId, ptyId]) => !referencesBinding({ ptyId, targetId, worktreeId, tabId, leafId }) ) ) if (Object.keys(nextBindings).length !== Object.keys(bindings).length) { diff --git a/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.test.ts b/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.test.ts index 97f10729ca0..d41b973f3ab 100644 --- a/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.test.ts +++ b/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.test.ts @@ -168,4 +168,45 @@ describe('workspace session terminal binding replay', () => { [LEAF_TWO]: 'pty-b' }) }) + + it('fails closed when one tab id is duplicated inside a single worktree list', () => { + // Map indexing is last-wins where a linear find was first-wins; the ambiguity + // fence must skip these ids so the two strategies can never disagree. + const prior = session(null) + prior.tabsByWorktree = { + [WORKTREE_A]: [ + terminalTab(WORKTREE_A, 'duplicate-tab', 'pty-first'), + terminalTab(WORKTREE_A, 'duplicate-tab', 'pty-last') + ] + } + const incoming = session(null) + incoming.tabsByWorktree = { + [WORKTREE_A]: [terminalTab(WORKTREE_A, 'duplicate-tab', null)] + } + + preserveMissingWorkspaceSessionTerminalBindings(incoming, prior, bindingRecovery as never) + + expect(incoming.tabsByWorktree[WORKTREE_A]![0]!.ptyId).toBeNull() + }) + + it('indexes prior tabs once when replaying a large workspace snapshot', () => { + let reads = 0 + const prior = session(null) + prior.tabsByWorktree.worktree = Array.from({ length: 1000 }, (_, i) => ({ + ...terminalTab('worktree', `tab-${i}`, `pty-${i}`), + get id() { + reads++ + return `tab-${i}` + } + })) + const incoming = session(null) + incoming.tabsByWorktree.worktree = Array.from({ length: 1000 }, (_, i) => + terminalTab('worktree', `tab-${i}`, null) + ) + preserveMissingWorkspaceSessionTerminalBindings(incoming, prior, bindingRecovery as never) + expect(reads).toBeLessThan(10_000) + expect(incoming.tabsByWorktree.worktree.map((tab) => tab.ptyId)).toEqual( + Array.from({ length: 1000 }, (_, i) => `pty-${i}`) + ) + }) }) diff --git a/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.ts b/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.ts index c064877b1d8..8e805b67b50 100644 --- a/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.ts +++ b/src/main/persistence/loading-store/workspace-session-terminal-binding-replay.ts @@ -93,6 +93,7 @@ export function preserveMissingWorkspaceSessionTerminalBindings( if (!priorList) { continue } + let priorById: Map | undefined for (const tab of tabs) { if (ambiguousTabIds.has(tab.id)) { continue @@ -100,7 +101,8 @@ export function preserveMissingWorkspaceSessionTerminalBindings( if (tab.ptyId) { continue } - const priorTab = priorList.find((candidate) => candidate.id === tab.id) + priorById ??= new Map(priorList.map((candidate) => [candidate.id, candidate])) + const priorTab = priorById.get(tab.id) const incomingLayout = nextLayouts[tab.id] const priorLayout = priorLayouts[tab.id] const priorPtyLeafId = priorLayout diff --git a/src/main/persistence/restoring-sessions/pane-alias-normalization-scaling.test.ts b/src/main/persistence/restoring-sessions/pane-alias-normalization-scaling.test.ts new file mode 100644 index 00000000000..2077af8d17a --- /dev/null +++ b/src/main/persistence/restoring-sessions/pane-alias-normalization-scaling.test.ts @@ -0,0 +1,114 @@ +import { expect, it, vi } from 'vitest' +import type { MigrationUnsupportedPtyEntry } from '../../../shared/agent-status-types' +import { legacyMigrationUnsupportedRowsToAliasEntries } from './pane-alias-normalization' + +vi.mock('../../agent-hooks/server', () => ({ agentHookServer: {} })) + +it('retains only the ambiguity verdict when many legacy rows name the same tab', () => { + const row: MigrationUnsupportedPtyEntry = { + ptyId: 'pty', + tabId: 'tab', + paneKey: 'tab:11111111-1111-4111-8111-111111111111', + reason: 'legacy-numeric-pane-key', + source: 'local', + updatedAt: 1 + } + const rows = Array.from({ length: 2000 }, () => row) + const iterator = Array.prototype[Symbol.iterator] + let copied = 0 + Array.prototype[Symbol.iterator] = function (this: unknown[]) { + if (this[0] === row) { + copied += this.length + } + return iterator.call(this) + } + let aliases: ReturnType + try { + aliases = legacyMigrationUnsupportedRowsToAliasEntries(rows) + } finally { + Array.prototype[Symbol.iterator] = iterator + } + expect(copied).toBeLessThan(10_000) + expect(aliases).toEqual([]) + const unique = legacyMigrationUnsupportedRowsToAliasEntries([row]) + expect(unique.map((entry) => entry.legacyPaneKey)).toEqual(['tab:0', 'tab:1']) + expect(unique.every((entry) => entry.stablePaneKey === row.paneKey)).toBe(true) +}) + +function legacyRow(overrides: Partial): MigrationUnsupportedPtyEntry { + return { + ptyId: 'pty', + tabId: 'tab', + paneKey: 'tab:11111111-1111-4111-8111-111111111111', + reason: 'legacy-numeric-pane-key', + source: 'local', + updatedAt: 1, + ...overrides + } +} + +// Ambiguity must fail closed: only an exactly-one row per tab may mint an alias. +it.each([ + ['0 rows', 0], + ['2 rows', 2], + ['3 rows', 3], + ['4 rows', 4] +])('mints no alias for a tab named by %s', (_label, count) => { + const rows = Array.from({ length: count }, (_, i) => + legacyRow({ + ptyId: `pty-${i}`, + paneKey: `tab:1111111${i}-1111-4111-8111-111111111111` + }) + ) + expect(legacyMigrationUnsupportedRowsToAliasEntries(rows)).toEqual([]) +}) + +it('mints both numeric aliases for a tab named by exactly 1 row', () => { + const row = legacyRow({ ptyId: 'pty-solo' }) + expect(legacyMigrationUnsupportedRowsToAliasEntries([row])).toEqual([ + { + ptyId: 'pty-solo', + legacyPaneKey: 'tab:0', + stablePaneKey: row.paneKey, + updatedAt: 1 + }, + { + ptyId: 'pty-solo', + legacyPaneKey: 'tab:1', + stablePaneKey: row.paneKey, + updatedAt: 1 + } + ]) +}) + +it('keeps unambiguous tabs in first-seen order while dropping ambiguous neighbours', () => { + const solo = legacyRow({ + tabId: 'solo', + ptyId: 'pty-solo', + paneKey: 'solo:11111111-1111-4111-8111-111111111111' + }) + const dupA = legacyRow({ + tabId: 'dup', + ptyId: 'pty-a', + paneKey: 'dup:22222222-2222-4222-8222-222222222222' + }) + const dupB = legacyRow({ + tabId: 'dup', + ptyId: 'pty-b', + paneKey: 'dup:33333333-3333-4333-8333-333333333333' + }) + const late = legacyRow({ + tabId: 'late', + ptyId: 'pty-late', + paneKey: 'late:44444444-4444-4444-8444-444444444444' + }) + // A third row for 'dup' must not resurrect it: ambiguity is sticky, not a parity toggle. + const aliases = legacyMigrationUnsupportedRowsToAliasEntries([solo, dupA, dupB, late, dupA]) + expect(aliases.map((entry) => entry.legacyPaneKey)).toEqual([ + 'solo:0', + 'solo:1', + 'late:0', + 'late:1' + ]) + expect(aliases.every((entry) => entry.ptyId !== 'pty-a' && entry.ptyId !== 'pty-b')).toBe(true) +}) diff --git a/src/main/persistence/restoring-sessions/pane-alias-normalization.ts b/src/main/persistence/restoring-sessions/pane-alias-normalization.ts index 057780ce143..c7e88c48021 100644 --- a/src/main/persistence/restoring-sessions/pane-alias-normalization.ts +++ b/src/main/persistence/restoring-sessions/pane-alias-normalization.ts @@ -14,21 +14,17 @@ export function legacyMigrationUnsupportedRowsToAliasEntries( const normalizedEntries = normalizeMigrationUnsupportedPtyEntries(entries).filter( (entry) => entry.tabId && entry.paneKey && parsePaneKey(entry.paneKey) ) - const entriesByTabId = new Map() + const entriesByTabId = new Map() for (const entry of normalizedEntries) { const tabId = entry.tabId if (!tabId) { continue } - entriesByTabId.set(tabId, [...(entriesByTabId.get(tabId) ?? []), entry]) + entriesByTabId.set(tabId, entriesByTabId.has(tabId) ? null : entry) } const aliasEntries: LegacyPaneKeyAliasEntry[] = [] - for (const [tabId, tabEntries] of entriesByTabId) { - if (tabEntries.length !== 1) { - continue - } - const [entry] = tabEntries - if (!entry.paneKey) { + for (const [tabId, entry] of entriesByTabId) { + if (!entry?.paneKey) { continue } // Why: pre-stable rows lack the old numeric key; only synthesize single-pane aliases when the row is unambiguous. diff --git a/src/main/pi/agent-status-extension-source.test.ts b/src/main/pi/agent-status-extension-source.test.ts index fed179837db..fa9823d76dd 100644 --- a/src/main/pi/agent-status-extension-source.test.ts +++ b/src/main/pi/agent-status-extension-source.test.ts @@ -481,12 +481,14 @@ describe('getPiAgentStatusExtensionSource', () => { await handlerCall }) - it('leaves runtime shutdown to PTY teardown instead of reporting turn completion', () => { + it('leaves runtime shutdown to PTY teardown instead of reporting turn completion', async () => { const harness = createHarness({ kind: 'pi' }) - // Why: Pi emits session_shutdown for reload/new/resume/fork while its PTY - // stays alive. agent_end is the only extension event that proves done. - expect(harness.handlers.session_shutdown).toBeUndefined() + // Why: Pi emits session_shutdown for reload/new/resume/fork while its PTY stays + // alive. agent_end is the only extension event that proves done, so the handler + // exists solely to release a dialog Pi tore down without a close. + await harness.callHook('session_shutdown') + expect(harness.fetchMock).not.toHaveBeenCalled() }) it('bounds stalled delivery to one active request and the latest pending status', async () => { diff --git a/src/main/pi/agent-status-extension-source.ts b/src/main/pi/agent-status-extension-source.ts index 6adfdf23fbc..38775ca1973 100644 --- a/src/main/pi/agent-status-extension-source.ts +++ b/src/main/pi/agent-status-extension-source.ts @@ -101,6 +101,7 @@ export function getPiAgentStatusExtensionSource(kind: PiAgentKind = 'pi'): strin '// Orca receiver from building an unbounded queue of obsolete snapshots.', 'const HOOK_POST_TIMEOUT_MS = 1000', 'let activePost = false', + ...(kind === 'pi' ? ['let piUiPromptDepth = 0', 'let piTurnInFlight = false'] : []), 'let pendingPost: { hookEventName: string; extra: Record; metadata: Record; ompRuntime: boolean } | null = null', ...sessionMetadataSourceLines, '', @@ -164,7 +165,10 @@ export function getPiAgentStatusExtensionSource(kind: PiAgentKind = 'pi'): strin ' const ompRuntime = isOmpRuntime()', ' pendingPost = {', ' hookEventName,', - ' extra,', + // Why: every coalesced snapshot must retain an open modal, not just its start event. + kind === 'pi' + ? ' extra: { ...extra, ...(!ompRuntime && piUiPromptDepth > 0 ? { ui_prompt_active: true } : {}) },' + : ' extra,', ' metadata: getPostSessionMetadata(ompRuntime),', ' ompRuntime,', ' }', diff --git a/src/main/pi/agent-status-handler-source.ts b/src/main/pi/agent-status-handler-source.ts index f5d7ef7eb01..9d02abbd78d 100644 --- a/src/main/pi/agent-status-handler-source.ts +++ b/src/main/pi/agent-status-handler-source.ts @@ -1,4 +1,5 @@ import type { PiAgentKind } from '../../shared/pi-agent-kind' +import { getPiAgentStatusUiPromptHandlerSourceLines } from './agent-status-ui-prompt-source' // Why: keep the generated handler registrations separate from hook transport; // both are independently sizeable and the installed extension concatenates them. @@ -8,6 +9,7 @@ export function getPiAgentStatusHandlerSourceLines(kind: PiAgentKind): string[] ? [ " pi.on('session_start', (event, ctx) => {", ' updateSessionMetadata(ctx)', + ...(kind === 'pi' ? [' piUiPromptDepth = 0'] : []), ' // Why: /reload re-registers the active session, but it is not a', ' // turn boundary and must not clear the visible status or unread state.', " if (event.reason === 'reload') return", @@ -104,6 +106,9 @@ export function getPiAgentStatusHandlerSourceLines(kind: PiAgentKind): string[] ...captureSessionMetadata, ' clearPendingAgentEndCheck()', ' agentEndReported = false', + // Why: a turn cannot begin under a dialog holding input focus, so this is the one + // boundary that can recover a modal whose close never arrived. + ...(kind === 'pi' ? [' piUiPromptDepth = 0', ' piTurnInFlight = true'] : []), " post('agent_start')", ' })', '', @@ -131,6 +136,7 @@ export function getPiAgentStatusHandlerSourceLines(kind: PiAgentKind): string[] ' })', '', ...approvalHandlers, + ...getPiAgentStatusUiPromptHandlerSourceLines(kind), " // Why: capture the assistant's final text on each completed message", ' // so the dashboard preview reflects the most recent reply even before', ' // agent_end fires. message_end is the right hook because pi guarantees', @@ -166,6 +172,9 @@ export function getPiAgentStatusHandlerSourceLines(kind: PiAgentKind): string[] ' function postAgentEndOnce(): void {', ' if (agentEndReported) return', ' agentEndReported = true', + // Why: distinct from agentEndReported, which also dedupes the completion post and so + // starts false on a pane that has not run a turn yet — that pane is idle, not busy. + ...(kind === 'pi' ? [' piTurnInFlight = false'] : []), " post('agent_end')", ' }', '', diff --git a/src/main/pi/agent-status-runtime-detection-source.ts b/src/main/pi/agent-status-runtime-detection-source.ts index 5d9cdbf6de8..6ba5edb69b9 100644 --- a/src/main/pi/agent-status-runtime-detection-source.ts +++ b/src/main/pi/agent-status-runtime-detection-source.ts @@ -1,26 +1,15 @@ import type { PiAgentKind } from '../../shared/pi-agent-kind' -export function getPiAgentStatusRuntimeDetectionSourceLines(kind: PiAgentKind): string[] { - if (kind === 'prime-agent') { - return [ - `const CONFIGURED_HOOK_PATH = '/hook/${kind}'`, - '', - 'function isOmpRuntime(): boolean {', - ' return false', - '}', - '', - 'function resolveHookPath(_ompRuntime: boolean): string {', - ' return CONFIGURED_HOOK_PATH', - '}' - ] - } - +/** Why: a bare-shell OMP launch runs inside a pi-kind pane, so every extension that has to + * defer to OMP's own approval events needs this check — not just the status extension it + * was first written for. */ +export function getPiOmpRuntimeDetectionSourceLines(configuredHookPath: string): string[] { return [ 'function processName(value: unknown): string {', " return String(value || '').split(/[\\\\/]/).pop()?.toLowerCase() || ''", '}', '', - `const CONFIGURED_HOOK_PATH = '/hook/${kind}'`, + `const CONFIGURED_HOOK_PATH = '${configuredHookPath}'`, 'let cachedOmpRuntime: boolean | null = null', '', 'function isOmpRuntime(): boolean {', @@ -39,7 +28,27 @@ export function getPiAgentStatusRuntimeDetectionSourceLines(kind: PiAgentKind): " ['omp', 'omp.js', 'omp.sh', 'omp.cmd', 'omp.exe', 'omp.bat'].includes(name)", ' )', ' return cachedOmpRuntime', - '}', + '}' + ] +} + +export function getPiAgentStatusRuntimeDetectionSourceLines(kind: PiAgentKind): string[] { + if (kind === 'prime-agent') { + return [ + `const CONFIGURED_HOOK_PATH = '/hook/${kind}'`, + '', + 'function isOmpRuntime(): boolean {', + ' return false', + '}', + '', + 'function resolveHookPath(_ompRuntime: boolean): string {', + ' return CONFIGURED_HOOK_PATH', + '}' + ] + } + + return [ + ...getPiOmpRuntimeDetectionSourceLines(`/hook/${kind}`), '', 'function resolveHookPath(ompRuntime: boolean): string {', ' // Why: runtime detection keeps a bare-shell OMP launch from reporting as Pi.', diff --git a/src/main/pi/agent-status-ui-prompt-source.ts b/src/main/pi/agent-status-ui-prompt-source.ts new file mode 100644 index 00000000000..5790c5c30a7 --- /dev/null +++ b/src/main/pi/agent-status-ui-prompt-source.ts @@ -0,0 +1,44 @@ +import type { PiAgentKind } from '../../shared/pi-agent-kind' + +/** Mirrors the titlebar extension's dialog tracking so both agree on when the wait ends. */ +export function getPiAgentStatusUiPromptHandlerSourceLines(kind: PiAgentKind): string[] { + if (kind !== 'pi') { + return [] + } + + return [ + " pi.on('ui_prompt_start', () => {", + ' if (isOmpRuntime()) return', + ' piUiPromptDepth++', + ' if (piUiPromptDepth > 1) return', + " post('ui_prompt_start')", + ' })', + '', + " pi.on('ui_prompt_end', (_event, ctx) => {", + ' if (isOmpRuntime() || piUiPromptDepth === 0) return', + ' piUiPromptDepth--', + ' if (piUiPromptDepth > 0) return', + ' // Why: ctx.isIdle throws outright once a session-switching modal invalidates the', + ' // runner (it calls assertActive), so local turn state is the floor, not a fallback:', + ' // with no turn in flight, no later event is coming to correct a working verdict, so', + ' // only consult ctx when this process believes work is running.', + ' let isIdle = !piTurnInFlight', + ' try {', + " if (!isIdle && typeof ctx?.isIdle === 'function') isIdle = ctx.isIdle() === true", + ' } catch {', + ' // Why: a runner this very modal invalidated cannot answer; keep the local verdict.', + ' }', + " post('ui_prompt_end', { is_idle: isIdle })", + ' })', + '', + " pi.on('session_shutdown', () => {", + ' if (isOmpRuntime()) return', + ' // Why: pi tears an open dialog down through resetExtensionUI without resolving its', + ' // promise, so a replaced session never emits the matching ui_prompt_end and the wait', + ' // would stick forever. Reset without posting: shutdown is not a turn boundary, and', + ' // the session_start that follows republishes the corrected state.', + ' piUiPromptDepth = 0', + ' })', + '' + ] +} diff --git a/src/main/pi/agent-status-ui-prompt.test.ts b/src/main/pi/agent-status-ui-prompt.test.ts new file mode 100644 index 00000000000..d91ccc37b34 --- /dev/null +++ b/src/main/pi/agent-status-ui-prompt.test.ts @@ -0,0 +1,326 @@ +import { describe, expect, it } from 'vitest' +import { normalizeHookPayload } from '../../shared/agent-hook-listener' +import { PANE_KEY } from '../../shared/agent-hook-listener-test-harness' +import { createHookListenerState } from '../../shared/agent-hook-listener/listener-state' +import { createAgentStatusExtensionHarness } from './agent-status-extension-test-harness' + +const HOOK_ENV = { ORCA_PANE_KEY: PANE_KEY, ORCA_AGENT_HOOK_ENV: 'production' } + +function createHarness() { + const state = createHookListenerState() + const statuses: ReturnType[] = [] + const harness = createAgentStatusExtensionHarness({ + kind: 'pi', + env: HOOK_ENV, + fetchImpl: async (_url, init) => { + statuses.push(normalizeHookPayload(state, 'pi', JSON.parse(String(init?.body)), 'production')) + return { ok: true } + } + }) + return { ...harness, statuses } +} + +async function flushPosts(): Promise { + // Each delivery has a bounded promise chain; no wall-clock sleeps in the harness. + for (let i = 0; i < 20; i++) { + await Promise.resolve() + } +} + +async function post(harness: ReturnType, name: string, event = {}) { + await harness.callHook(name, event) + await flushPosts() +} + +describe('Pi UI prompt status', () => { + it.each(['select', 'confirm', 'input', 'editor', 'custom'])( + 'blocks for %s without exposing modal contents and resumes work on close', + async (kind) => { + const harness = createHarness() + await post(harness, 'agent_start') + await post(harness, 'ui_prompt_start', { kind, title: 'Private title' }) + expect(harness.statuses.at(-1)?.payload.state).toBe('waiting') + const body = JSON.parse(String(harness.fetchMock.mock.calls.at(-1)?.[1]?.body)) + expect(body.payload).toEqual({ hook_event_name: 'ui_prompt_start', ui_prompt_active: true }) + + await harness.callHook('ui_prompt_end', { kind }, { isIdle: () => false }) + await flushPosts() + expect(harness.statuses.at(-1)?.payload.state).toBe('working') + } + ) + + it.each([ + 'tool_call', + 'tool_execution_start', + 'tool_execution_end', + 'message_end', + 'agent_settled' + ])('%s cannot clear an open modal', async (name) => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await post(harness, name, { + toolName: 'ask_user_question', + input: { questions: [{ question: 'Stale question' }] }, + message: { role: 'assistant', content: [{ type: 'text', text: 'Still here' }] } + }) + expect(harness.statuses.at(-1)?.payload).toMatchObject({ state: 'waiting', agentType: 'pi' }) + expect(harness.statuses.at(-1)?.payload.toolName).toBeUndefined() + expect(harness.statuses.at(-1)?.payload.interactivePrompt).toBeUndefined() + await harness.callHook('ui_prompt_end', {}, { isIdle: () => true }) + await flushPosts() + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('clears stale question cards when a generic modal opens', async () => { + const harness = createHarness() + await post(harness, 'tool_call', { + toolName: 'ask_user_question', + input: { questions: [{ question: 'Pick one' }] } + }) + expect(harness.statuses.at(-1)?.payload.interactivePrompt).toBeDefined() + await post(harness, 'ui_prompt_start') + expect(harness.statuses.at(-1)?.payload.toolName).toBeUndefined() + expect(harness.statuses.at(-1)?.payload.toolInput).toBeUndefined() + expect(harness.statuses.at(-1)?.payload.interactivePrompt).toBeUndefined() + }) + + it('returns an idle session to done after its modal closes', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await harness.callHook('ui_prompt_end', {}, { isIdle: () => true }) + await flushPosts() + expect(harness.statuses.map((status) => status?.payload.state)).toEqual(['waiting', 'done']) + }) + + it('returns a pane that never ran a turn to done when idleness is unreadable', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await post(harness, 'ui_prompt_end') + // Why: no turn has started, so the pane is idle — reporting working would spin forever. + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('trusts local turn state over a ctx that claims work on an idle pane', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await harness.callHook('ui_prompt_end', {}, { isIdle: () => false }) + await flushPosts() + // Why: no turn ever started, so nothing later would correct a working verdict. + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('lets the normal settlement hook finish work after a modal closes', async () => { + const harness = createHarness() + await post(harness, 'agent_start') + await post(harness, 'ui_prompt_start') + await harness.callHook('ui_prompt_end', {}, { isIdle: () => false }) + await flushPosts() + await post(harness, 'agent_settled') + expect(harness.statuses.map((status) => status?.payload.state)).toEqual([ + 'working', + 'waiting', + 'working', + 'done' + ]) + }) + + it('retains modal state across an in-process registration reload', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + harness.reload() + await post(harness, 'tool_execution_end', { toolName: 'bash' }) + // Why: re-registering handlers is not a session boundary and must not lose the wait. + expect(harness.statuses.at(-1)?.payload.state).toBe('waiting') + }) + + it('releases a modal that a session replacement tore down without a close', async () => { + const harness = createHarness() + await post(harness, 'before_agent_start', { prompt: 'Old session prompt' }) + await post(harness, 'ui_prompt_start') + expect(harness.statuses.at(-1)?.payload.state).toBe('waiting') + // Why: pi hides the dialog through resetExtensionUI without resolving its promise, + // so no ui_prompt_end is ever emitted — these two boundaries are the only release. + await post(harness, 'session_shutdown') + await post(harness, 'session_start', { reason: 'switch' }) + await post(harness, 'tool_execution_end', { toolName: 'bash' }) + expect(harness.statuses.at(-1)?.payload.state).not.toBe('waiting') + }) + + it('releases a modal dropped by a reload that emits no shutdown', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await post(harness, 'session_start', { reason: 'reload' }) + await post(harness, 'tool_execution_end', { toolName: 'bash' }) + expect(harness.statuses.at(-1)?.payload.state).not.toBe('waiting') + }) + + it('still captures the assistant reply that lands while a modal is open', async () => { + const harness = createHarness() + await post(harness, 'agent_start') + await post(harness, 'message_end', { + message: { role: 'assistant', content: [{ type: 'text', text: 'Before modal' }] } + }) + await post(harness, 'ui_prompt_start') + await post(harness, 'message_end', { + message: { role: 'assistant', content: [{ type: 'text', text: 'Final reply' }] } + }) + await harness.callHook('ui_prompt_end', {}, { isIdle: () => true }) + await flushPosts() + expect(harness.statuses.at(-1)?.payload).toMatchObject({ + state: 'done', + lastAssistantMessage: 'Final reply' + }) + expect(harness.statuses.at(-1)?.payload.toolName).toBeUndefined() + expect(harness.statuses.at(-1)?.payload.interactivePrompt).toBeUndefined() + }) + + it('still reports the close when the modal invalidated its own runner', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await harness.callHook( + 'ui_prompt_end', + {}, + { + isIdle: () => { + throw new Error('extension runner is no longer active') + } + } + ) + await flushPosts() + // Why: a lost close would strand the pane on waiting; no turn is running, so done. + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('keeps a mid-turn modal working when its runner throws on close', async () => { + const harness = createHarness() + await post(harness, 'agent_start') + await post(harness, 'ui_prompt_start') + await harness.callHook( + 'ui_prompt_end', + {}, + { + isIdle: () => { + throw new Error('extension runner is no longer active') + } + } + ) + await flushPosts() + // Why: the turn is still in flight, so done would ring the completion bell early. + expect(harness.statuses.at(-1)?.payload.state).toBe('working') + await post(harness, 'agent_settled') + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('recovers on a new turn when a modal close was lost', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + expect(harness.statuses.at(-1)?.payload.state).toBe('waiting') + // Why: a turn cannot begin under a dialog holding input focus, so this is recovery. + await post(harness, 'agent_start') + await post(harness, 'tool_execution_end', { toolName: 'bash' }) + expect(harness.statuses.at(-1)?.payload.state).toBe('working') + }) + + it('keeps the wait until the outermost of nested modals closes', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_start') + await post(harness, 'ui_prompt_start') + await harness.callHook('ui_prompt_end', {}, { isIdle: () => true }) + await flushPosts() + expect(harness.statuses.at(-1)?.payload.state).toBe('waiting') + await harness.callHook('ui_prompt_end', {}, { isIdle: () => true }) + await flushPosts() + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('returns an idle pane to done when its modal lost the runner', async () => { + const harness = createHarness() + await post(harness, 'agent_start') + await post(harness, 'agent_settled') + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + await post(harness, 'ui_prompt_start') + await harness.callHook( + 'ui_prompt_end', + {}, + { + isIdle: () => { + throw new Error('extension runner is no longer active') + } + } + ) + await flushPosts() + // Why: the turn already reported its end, so no later event is coming to correct a + // guess of working — fall back to what this process knows rather than strand it. + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('keeps a mid-turn modal working when its close cannot read idleness', async () => { + const harness = createHarness() + await post(harness, 'agent_start') + await post(harness, 'ui_prompt_start') + await post(harness, 'ui_prompt_end') + expect(harness.statuses.at(-1)?.payload.state).toBe('working') + await post(harness, 'agent_settled') + expect(harness.statuses.at(-1)?.payload.state).toBe('done') + }) + + it('ignores an unmatched prompt end', async () => { + const harness = createHarness() + await post(harness, 'ui_prompt_end') + expect(harness.fetchMock).not.toHaveBeenCalled() + }) + + it('isolates prompt state between Pi processes', async () => { + const first = createHarness() + const second = createHarness() + await post(first, 'ui_prompt_start') + await post(second, 'agent_start') + expect(first.statuses.at(-1)?.payload.state).toBe('waiting') + expect(second.statuses.at(-1)?.payload.state).toBe('working') + }) + + it('preserves blocked when a stalled sender coalesces away the start event', async () => { + let finish: (() => void) | undefined + const harness = createAgentStatusExtensionHarness({ + kind: 'pi', + env: HOOK_ENV, + fetchImpl: () => + new Promise((resolve) => { + finish = resolve + }) + }) + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + await harness.callHook('tool_execution_end', { toolName: 'bash' }) + expect(harness.fetchMock).toHaveBeenCalledTimes(1) + finish?.() + await flushPosts() + const state = createHookListenerState() + const latest = JSON.parse(String(harness.fetchMock.mock.calls.at(-1)?.[1]?.body)) + expect(latest.payload.hook_event_name).toBe('tool_execution_end') + expect(normalizeHookPayload(state, 'pi', latest, 'production')?.payload.state).toBe('waiting') + + await harness.callHook('ui_prompt_end', {}, { isIdle: () => false }) + await harness.callHook('tool_execution_start', { toolName: 'bash', args: { command: 'pwd' } }) + finish?.() + await flushPosts() + const resumed = JSON.parse(String(harness.fetchMock.mock.calls.at(-1)?.[1]?.body)) + expect(normalizeHookPayload(state, 'pi', resumed, 'production')?.payload.state).toBe('working') + finish?.() + await flushPosts() + }) + + it.each([ + { kind: 'omp' as const }, + { kind: 'prime-agent' as const }, + { kind: 'pi' as const, title: 'omp' } + ])('does not add Pi prompt status to $kind ($title)', async (args) => { + const harness = createAgentStatusExtensionHarness(args) + await harness.callHook('ui_prompt_start') + await harness.callHook('ui_prompt_end', {}, { isIdle: () => true }) + expect(harness.fetchMock).not.toHaveBeenCalled() + await harness.callHook('tool_call', { toolName: 'bash', input: { command: 'pwd' } }) + const body = JSON.parse(String(harness.fetchMock.mock.calls[0]?.[1]?.body)) + expect(body.payload.ui_prompt_active).toBeUndefined() + }) +}) diff --git a/src/main/pi/titlebar-extension-service.ts b/src/main/pi/titlebar-extension-service.ts index 3a43ce4ac38..8a093096f4e 100644 --- a/src/main/pi/titlebar-extension-service.ts +++ b/src/main/pi/titlebar-extension-service.ts @@ -150,7 +150,7 @@ export class PiTitlebarExtensionService { if (kind !== 'prime-agent') { this.writeManagedExtension( join(extensionsDir, ORCA_PI_EXTENSION_FILE), - withOrcaManagedExtensionMarker(getPiTitlebarExtensionSource()) + withOrcaManagedExtensionMarker(getPiTitlebarExtensionSource(kind)) ) this.writeManagedExtension( join(extensionsDir, ORCA_PI_PREFILL_EXTENSION_FILE), diff --git a/src/main/pi/titlebar-extension-source.test.ts b/src/main/pi/titlebar-extension-source.test.ts index bee2e007c57..be21f8c6a16 100644 --- a/src/main/pi/titlebar-extension-source.test.ts +++ b/src/main/pi/titlebar-extension-source.test.ts @@ -3,6 +3,8 @@ import { runInNewContext } from 'node:vm' import ts from 'typescript-api' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { detectAgentStatusFromTitle } from '../../shared/agent-detection' +import type { PiAgentKind } from '../../shared/pi-agent-kind' import { getPiTitlebarExtensionSource } from './titlebar-extension-source' const BRAILLE_RE = /[⠀-⣿]/ @@ -23,8 +25,19 @@ type Harness = { const CWD = '/repo/orca-app' const SESSION = 'omp-session' const IDLE_TITLE = `π - ${SESSION} - orca-app` +const PROMPT_TITLE = `π ! ${SESSION} - orca-app` -function createHarness(options: { paneKey?: string; isIdle?: () => boolean } = {}): Harness { +function createHarness( + options: { + paneKey?: string + isIdle?: () => boolean + kind?: PiAgentKind + processTitle?: string + cwdImpl?: () => string + sessionNameImpl?: () => string + env?: Record + } = {} +): Harness { const titles: string[] = [] const ctx: TitlebarContext = { ui: { @@ -48,8 +61,11 @@ function createHarness(options: { paneKey?: string; isIdle?: () => boolean } = { module, exports: module.exports, process: { - env: { ORCA_PANE_KEY: options.paneKey ?? 'pane-1' }, - cwd: () => CWD + env: { ORCA_PANE_KEY: options.paneKey ?? 'pane-1', ...options.env }, + pid: options.env?.ORCA_PI_TITLE_MARKER_OWNED === undefined ? 111 : 222, + title: options.processTitle ?? 'pi', + argv: ['node', 'pi'], + cwd: options.cwdImpl ?? (() => CWD) }, console: { warn: vi.fn(), error: vi.fn(), log: vi.fn() }, Promise, @@ -61,7 +77,7 @@ function createHarness(options: { paneKey?: string; isIdle?: () => boolean } = { } as Record context.globalThis = context - const output = ts.transpileModule(getPiTitlebarExtensionSource(), { + const output = ts.transpileModule(getPiTitlebarExtensionSource(options.kind ?? 'pi'), { compilerOptions: { module: ts.ModuleKind.CommonJS, target: ts.ScriptTarget.ES2020 } }).outputText runInNewContext(output, context) @@ -76,7 +92,7 @@ function createHarness(options: { paneKey?: string; isIdle?: () => boolean } = { on(name: string, handler: HookHandler) { handlers[name] = handler }, - getSessionName: () => SESSION + getSessionName: options.sessionNameImpl ?? (() => SESSION) }) return { @@ -247,4 +263,329 @@ describe('getPiTitlebarExtensionSource', () => { expect(vi.getTimerCount()).toBe(0) expect(harness.lastTitle()).toBe(IDLE_TITLE) }) + + it('marks a mid-turn dialog as needing input and holds it against the spinner', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + expect(detectAgentStatusFromTitle(PROMPT_TITLE)).toBe('permission') + + // Why: the spinner interval keeps running, but must not repaint over the marker. + await vi.advanceTimersByTimeAsync(800) + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + await harness.callHook('ui_prompt_end') + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + expect(vi.getTimerCount()).toBe(1) + }) + + it('returns an idle pane to its plain title when the dialog closes', async () => { + const harness = createHarness() + + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + await harness.callHook('ui_prompt_end') + expect(harness.lastTitle()).toBe(IDLE_TITLE) + expect(vi.getTimerCount()).toBe(0) + }) + + it('only the outermost of nested dialogs moves the title', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + await harness.callHook('ui_prompt_start') + await harness.callHook('ui_prompt_end') + // Why: the outer dialog still holds input focus. + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + await harness.callHook('ui_prompt_end') + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('ignores an unmatched dialog close', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + const titleCount = harness.titles.length + await harness.callHook('ui_prompt_end') + expect(harness.titles.length).toBe(titleCount) + }) + + it.each(['agent_settled', 'session_shutdown'])( + 'keeps the marker when %s lands under an open dialog', + async (name) => { + const harness = createHarness() + + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + await harness.callHook(name) + // Why: settling does not answer the dialog, so the pane still needs the user. + const expected = name === 'session_shutdown' ? IDLE_TITLE : PROMPT_TITLE + expect(harness.lastTitle()).toBe(expected) + // Why: settling stops the spinner but must leave the marker re-assert running, or + // pi's own next title write would silently retire a dialog that is still open. + expect(vi.getTimerCount()).toBe(name === 'session_shutdown' ? 0 : 1) + } + ) + + it('keeps the marker across an idle compaction that finishes under a dialog', async () => { + const harness = createHarness() + + await harness.callHook('ui_prompt_start') + await harness.callHook('auto_compaction_start', { reason: 'idle' }) + await harness.callHook('auto_compaction_end') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + }) + + it('recovers the spinner on a new turn when a dialog close was lost', async () => { + const harness = createHarness() + + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + // Why: a turn cannot start under a dialog holding input focus, so this is recovery. + await harness.callHook('agent_start') + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('leaves the marker to OMP approval events instead of painting it', () => { + expect(createHarness({ kind: 'omp' }).handlers.ui_prompt_start).toBeUndefined() + }) + + it('still caps idle maintenance while a dialog holds the title', async () => { + const harness = createHarness() + + await harness.callHook('auto_compaction_start', { reason: 'idle' }) + await harness.callHook('ui_prompt_start') + // Why: an open dialog must not suspend the cap that stops a stranded spinner. + vi.advanceTimersByTime(301_000) + + // Why: the spinner is capped, but the marker re-assert survives it — the dialog is + // still open, so the pane must keep reporting that it needs input. + expect(vi.getTimerCount()).toBe(1) + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + }) + + it('survives a dialog event that carries no ui context', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + await expect(harness.handlers.ui_prompt_start?.({}, undefined)).resolves.toBeUndefined() + await expect(harness.handlers.ui_prompt_end?.({}, undefined)).resolves.toBeUndefined() + }) + + it('keeps spinning when the dialog event could not paint the marker', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + await harness.handlers.ui_prompt_start?.({}, undefined) + // Why: suppressing frames without a marker would freeze the title mid-spinner, which + // still reads as working — the opposite of what the marker is for. + await vi.advanceTimersByTimeAsync(160) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('marks a nested dialog when the outer one could not paint', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + await harness.handlers.ui_prompt_start?.({}, undefined) + await harness.callHook('ui_prompt_start') + // Why: the outer ctx cannot decide that the whole stack stays unmarked. + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + }) + + it('clears the marker through the opening ctx when the close carries none', async () => { + const harness = createHarness() + + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + await harness.handlers.ui_prompt_end?.({}, undefined) + // Why: otherwise the pane asks for attention until the next turn. + expect(harness.lastTitle()).toBe(IDLE_TITLE) + }) + + it('does not reject when the dialog ctx can no longer paint', async () => { + const harness = createHarness() + const throwing = { + ui: { + setTitle: () => { + throw new Error('extension runner is no longer active') + } + } + } + + await expect(harness.handlers.ui_prompt_start?.({}, throwing)).resolves.toBeUndefined() + // Why: the marker never went up, so the spinner must not stay suppressed. + await harness.callHook('agent_start') + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('does not reject when the captured ctx dies before the dialog closes', async () => { + const harness = createHarness() + let live = true + const dying = { + ui: { + setTitle: (title: string) => { + if (!live) { + throw new Error('extension runner is no longer active') + } + harness.titles.push(title) + } + } + } + + await harness.handlers.ui_prompt_start?.({}, dying) + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + live = false + // Why: the close carries no ui, so it falls back to the ctx the modal invalidated. + await expect(harness.handlers.ui_prompt_end?.({}, undefined)).resolves.toBeUndefined() + // Why: a later turn still recovers a clean title through a live ctx. + await harness.callHook('agent_start') + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('does not strand the marker when the closing ctx throws on ui access', async () => { + const harness = createHarness() + // Why: pi's ctx.ui is a getter that calls assertActive(); a session-replacing dialog + // invalidates the runner, so reading ctx.ui throws rather than yielding undefined. + const stale = { + get ui(): never { + throw new Error('This extension ctx is stale') + } + } + + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + await expect(harness.handlers.ui_prompt_end?.({}, stale as never)).resolves.toBeUndefined() + // Why: the opening ctx still paints, so the pane stops asking for input. + expect(harness.lastTitle()).toBe(IDLE_TITLE) + + // Why: a stranded markerPainted would suppress every later working frame. + await harness.callHook('agent_start') + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('does not reject when the opening ctx throws on ui access', async () => { + const harness = createHarness() + const stale = { + get ui(): never { + throw new Error('This extension ctx is stale') + } + } + + await harness.callHook('agent_start') + await expect(harness.handlers.ui_prompt_start?.({}, stale as never)).resolves.toBeUndefined() + // Why: no marker went up, so the spinner must keep running. + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('re-asserts the marker when pi repaints the title under a dialog', async () => { + const harness = createHarness() + + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + // Why: pi repaints on session_info_changed/rebindCurrentSession with no event we see, + // so a marker that is merely "not overwritten by us" would be silently lost. + harness.titles.push('π - other - orca-app') + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + }) + + it('re-asserts the marker on an idle pane with no spinner running', async () => { + const harness = createHarness() + + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + // Why: no turn is running, so renderFrame never fires — only the slow re-assert can + // undo a title pi writes from session_info_changed or its update-check restore. + harness.titles.push('\u03c0 - other - orca-app') + await vi.advanceTimersByTimeAsync(1000) + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + await harness.callHook('ui_prompt_end') + expect(harness.lastTitle()).toBe(IDLE_TITLE) + expect(vi.getTimerCount()).toBe(0) + }) + + it('releases the marker when a session replacement drops the dialog', async () => { + const harness = createHarness() + + await harness.callHook('ui_prompt_start') + expect(harness.lastTitle()).toBe(PROMPT_TITLE) + + // Why: pi hides the dialog without resolving it, so no close is coming. + await harness.callHook('session_start', { reason: 'switch' }) + expect(vi.getTimerCount()).toBe(0) + await harness.callHook('agent_start') + await vi.advanceTimersByTimeAsync(80) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('survives a deleted cwd instead of crashing the pi process', async () => { + const harness = createHarness({ + cwdImpl: () => { + throw new Error('ENOENT: uv_cwd') + } + }) + + // Why: these run inside setInterval callbacks, where an escape is an uncaught + // exception and pi exits(1) through its own uncaughtException handler. + await expect(harness.callHook('agent_start')).resolves.toBeUndefined() + await expect(harness.callHook('ui_prompt_start')).resolves.toBeUndefined() + // Why: an unguarded throw in the interval would surface here as an unhandled error. + await vi.advanceTimersByTimeAsync(2000) + await expect(harness.callHook('ui_prompt_end')).resolves.toBeUndefined() + await expect(harness.callHook('agent_settled')).resolves.toBeUndefined() + }) + + it('survives a session name that throws on a stale runtime', async () => { + let live = true + const harness = createHarness({ + sessionNameImpl: () => { + if (!live) { + throw new Error('This extension API is stale') + } + return SESSION + } + }) + + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + live = false + await vi.advanceTimersByTimeAsync(2000) + await expect(harness.callHook('ui_prompt_end')).resolves.toBeUndefined() + }) + + it('leaves the needs-input marker to the process that owns the pane', async () => { + // Why: child agents inherit ORCA_PANE_KEY, and a second process asserting the marker + // would report needs-input for a pane it does not speak for. + const harness = createHarness({ env: { ORCA_PI_TITLE_MARKER_OWNED: '111' } }) + + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + await vi.advanceTimersByTimeAsync(1000) + expect(harness.titles).not.toContain(PROMPT_TITLE) + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + }) + + it('leaves an OMP runtime to its own approval events', () => { + const harness = createHarness({ processTitle: 'omp' }) + + expect(harness.handlers.ui_prompt_start).toBeDefined() + expect(() => harness.handlers.ui_prompt_start?.({}, undefined)).not.toThrow() + }) }) diff --git a/src/main/pi/titlebar-extension-source.ts b/src/main/pi/titlebar-extension-source.ts index a41eafb896f..7fc15c191bc 100644 --- a/src/main/pi/titlebar-extension-source.ts +++ b/src/main/pi/titlebar-extension-source.ts @@ -1,7 +1,54 @@ +import type { PiAgentKind } from '../../shared/pi-agent-kind' +import { getPiOmpRuntimeDetectionSourceLines } from './agent-status-runtime-detection-source' + export const ORCA_PI_EXTENSION_FILE = 'orca-titlebar-spinner.ts' -export function getPiTitlebarExtensionSource(): string { +export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { + // Why: OMP reports input waits through its own approval events, which the status + // extension already maps, and it writes this same marker natively. The runtime check + // matters as well as the kind: a bare-shell OMP launch runs inside a pi-kind pane. + const uiPromptHandlers = + kind === 'pi' + ? [ + " pi.on('ui_prompt_start', async (_event, ctx) => {", + ' if (isOmpRuntime() || !ownsMarker) return', + ' promptDepth++', + ' // Why: retry on every open rather than only the outermost, so an outer ctx', + ' // that could not paint cannot decide the whole stack stays unmarked.', + ' if (markerPainted) return', + ' const painter = resolvePainter(ctx)', + ' // Why: only hold the spinner off once the marker is actually up, or a ctx', + ' // that cannot paint would freeze the title on its last working frame.', + " if (!paintTitle(painter, () => getMarkedTitle(pi, '!'))) return", + ' markerPainted = true', + ' promptCtx = painter', + ' startMarkerReassert(painter)', + ' })', + '', + " pi.on('ui_prompt_end', async (_event, ctx) => {", + ' if (isOmpRuntime() || !ownsMarker || promptDepth === 0) return', + ' promptDepth--', + ' if (promptDepth > 0) return', + ' // Why: the opening ctx already painted once, so a close whose own ctx is stale', + ' // does not leave the needs-input marker up until the next turn.', + ' const painter = resolvePainter(ctx) ?? promptCtx', + ' markerPainted = false', + ' promptCtx = null', + ' stopMarkerReassert()', + ' // Why: a still-live turn resumes its spinner in place; otherwise the pane is idle', + ' // and must drop the needs-input marker rather than keep asking for attention.', + ' if (timer) {', + ' renderFrame(painter)', + ' return', + ' }', + ' paintTitle(painter, () => getBaseTitle(pi))', + ' })', + '' + ] + : [] + return [ + ...(kind === 'pi' ? [...getPiOmpRuntimeDetectionSourceLines(`/hook/${kind}`), ''] : []), 'const BRAILLE_FRAMES = [', " '\\u280b',", " '\\u2819',", @@ -16,36 +63,111 @@ export function getPiTitlebarExtensionSource(): string { ']', '', 'const FRAME_INTERVAL_MS = 80', + '// Why: pi repaints the title from its own writers (session_info_changed, the win32', + '// update-check restore) with no event we observe, so the marker has to be re-asserted', + '// even when no spinner frame is due. Coarse on purpose: it only rewrites one string.', + 'const MARKER_REASSERT_MS = 1000', 'const AGENT_END_IDLE_RECHECK_MS = 25', 'const AGENT_END_IDLE_RECHECK_MAX_MS = 250', '// Why: a failed idle compaction can end without auto_compaction_end, and no agent turn will', '// close a maintenance spinner — cap it so idle maintenance cannot strand a working title.', 'const IDLE_COMPACTION_MAX_FRAMES = Math.ceil(300000 / FRAME_INTERVAL_MS)', '', - 'function getBaseTitle(pi) {', + '// Why: `-` is the plain separator; `!` is the state marker Orca reads as needs-input', + '// (src/shared/pi-state-title-marker.ts), so mobile and the CLI see the wait too.', + 'function getMarkedTitle(pi, marker) {', ' const cwd = process.cwd().split(/[\\\\/]/).filter(Boolean).at(-1) || process.cwd()', ' const session = pi.getSessionName()', - ' return session ? `\\u03c0 - ${session} - ${cwd}` : `\\u03c0 - ${cwd}`', + ' return session', + ' ? `\\u03c0 ${marker} ${session} - ${cwd}`', + ' : `\\u03c0 ${marker} ${cwd}`', + '}', + '', + 'function getBaseTitle(pi) {', + " return getMarkedTitle(pi, '-')", + '}', + '', + '// Why: the ctx.ui pi passes is a getter that calls assertActive() and throws once a', + '// session-replacing dialog invalidates the runner; optional chaining cannot screen', + '// that out. Read it behind a try and never mutate state before a paint has succeeded.', + 'function resolvePainter(ctx) {', + ' try {', + " return typeof ctx?.ui?.setTitle === 'function' ? ctx : null", + ' } catch {', + ' return null', + ' }', + '}', + '', + '// Why: buildTitle runs inside the try because it is not safe either — getSessionName()', + '// calls assertActive() and process.cwd() throws ENOENT once the worktree is deleted.', + '// Most call sites are timer callbacks, where an escape is an uncaught exception and pi', + '// exits(1) through its own uncaughtException handler.', + 'function paintTitle(ctx, buildTitle) {', + ' if (!ctx) return false', + ' try {', + ' ctx.ui.setTitle(buildTitle())', + ' return true', + ' } catch {', + ' return false', + ' }', '}', '', 'export default function (pi) {', ' if (!process.env.ORCA_PANE_KEY) return', + ...(kind === 'pi' + ? [ + ' // Why: child agents inherit the pane env, and the spinner is harmlessly', + ' // per-process — but the needs-input marker is status the pane reports, so only', + ' // one process may assert it. Mirrors ORCA_PI_STATUS_OWNED in the status hook.', + ' const markerOwnerPid = process.env.ORCA_PI_TITLE_MARKER_OWNED', + ' const ownsMarker = !markerOwnerPid || markerOwnerPid === String(process.pid)', + ' if (ownsMarker) process.env.ORCA_PI_TITLE_MARKER_OWNED = String(process.pid)' + ] + : []), + ' let timer = null', ' let frameIndex = 0', ' // Why: only idle maintenance owns a spinner of its own. A threshold compaction runs', ' // inside an agent turn, whose spinner must outlive it, and any newer start clears the', ' // marker so a late idle completion cannot stop current work (#16470).', ' let idleCompactionOwnsSpinner = false', + ' // Why: pi already collapses nested prompts into one start/end pair, so this counter', + ' // guards a close that never arrives, not nesting. A new turn cannot start under a', + ' // dialog holding input focus, so agent_start doubles as recovery.', + ' let promptDepth = 0', + ' let markerPainted = false', + ' let promptCtx = null', + ' // Why: a separate handle from `timer`, which clearAnimation() nulls — the marker must', + ' // survive a turn settling, a shutdown of the spinner, and the idle-maintenance cap.', + ' let markerTimer = null', ' let pendingAgentEndCheck = null', ' let pendingAgentEndContext = null', ' let agentEndIdleRecheckMs = AGENT_END_IDLE_RECHECK_MS', '', + ' function resetPromptState() {', + ' stopMarkerReassert()', + ' promptDepth = 0', + ' markerPainted = false', + ' promptCtx = null', + ' }', + '', ' function clearPendingAgentEndCheck() {', ' if (pendingAgentEndCheck !== null) clearTimeout(pendingAgentEndCheck)', ' pendingAgentEndCheck = null', ' pendingAgentEndContext = null', ' }', '', + ' function stopMarkerReassert() {', + ' if (markerTimer) clearInterval(markerTimer)', + ' markerTimer = null', + ' }', + '', + ' function startMarkerReassert(ctx) {', + ' stopMarkerReassert()', + " markerTimer = setInterval(() => paintTitle(ctx, () => getMarkedTitle(pi, '!')), MARKER_REASSERT_MS)", + " if (typeof markerTimer.unref === 'function') markerTimer.unref()", + ' }', + '', ' function clearAnimation() {', ' if (timer) {', ' clearInterval(timer)', @@ -58,19 +180,35 @@ export function getPiTitlebarExtensionSource(): string { ' function stopAnimation(ctx) {', ' clearPendingAgentEndCheck()', ' clearAnimation()', - ' ctx.ui.setTitle(getBaseTitle(pi))', + ' // Why: settling under an open dialog still leaves the pane waiting on the user, so', + ' // the idle title must not retire the marker the dialog is holding.', + " paintTitle(ctx, () => (markerPainted ? getMarkedTitle(pi, '!') : getBaseTitle(pi)))", ' }', '', ' function renderFrame(ctx) {', + ' // Why: the maintenance cap runs before the dialog guard so a dialog left open', + ' // cannot suspend it; stopAnimation keeps the marker while a dialog is open.', ' if (idleCompactionOwnsSpinner && frameIndex >= IDLE_COMPACTION_MAX_FRAMES) {', ' stopAnimation(ctx)', ' return', ' }', - ' const frame = BRAILLE_FRAMES[frameIndex % BRAILLE_FRAMES.length]', - ' const cwd = process.cwd().split(/[\\\\/]/).filter(Boolean).at(-1) || process.cwd()', - ' const session = pi.getSessionName()', - ' const title = session ? `${frame} \\u03c0 - ${session} - ${cwd}` : `${frame} \\u03c0 - ${cwd}`', - ' ctx.ui.setTitle(title)', + ' // Why: an 80ms working frame would repaint over the needs-input marker within one', + ' // tick, so a mid-turn dialog would still look busy everywhere the title is the', + ' // only evidence. Re-assert rather than skip: pi repaints the title on its own', + ' // (session_info_changed, resetExtensionUI, rebindCurrentSession) and would', + ' // otherwise wipe the marker with nothing to restore it. The frame still counts,', + ' // so the cap above keeps accruing in wall-clock.', + ' if (markerPainted) {', + " paintTitle(ctx, () => getMarkedTitle(pi, '!'))", + ' frameIndex++', + ' return', + ' }', + ' paintTitle(ctx, () => {', + ' const frame = BRAILLE_FRAMES[frameIndex % BRAILLE_FRAMES.length]', + ' const cwd = process.cwd().split(/[\\\\/]/).filter(Boolean).at(-1) || process.cwd()', + ' const session = pi.getSessionName()', + ' return session ? `${frame} \\u03c0 - ${session} - ${cwd}` : `${frame} \\u03c0 - ${cwd}`', + ' })', ' frameIndex++', ' }', '', @@ -101,9 +239,17 @@ export function getPiTitlebarExtensionSource(): string { ' }', '', " pi.on('agent_start', async (_event, ctx) => {", + ' resetPromptState()', ' startAnimation(ctx)', ' })', '', + ' // Why: pi drops an open dialog through resetExtensionUI without resolving its promise,', + ' // so a replaced or reloaded session never sends the matching close. Both boundaries', + ' // prove no dialog from the old session is still on screen.', + " pi.on('session_start', async () => {", + ' resetPromptState()', + ' })', + '', ' // Why: modern Pi/OMP emit agent_end mid-run and only settle later, so settlement is the', ' // authoritative completion boundary. Legacy runtimes never emit it, so agent_end stays.', " pi.on('agent_settled', async (_event, ctx) => {", @@ -126,6 +272,7 @@ export function getPiTitlebarExtensionSource(): string { " if (typeof pendingAgentEndCheck.unref === 'function') pendingAgentEndCheck.unref()", ' })', '', + ...uiPromptHandlers, " pi.on('auto_compaction_start', async (event, ctx) => {", " if (event?.reason !== 'idle') return", ' // Why: the idle worker can fire against a turn that just started, and reason alone does', @@ -142,6 +289,7 @@ export function getPiTitlebarExtensionSource(): string { ' })', '', " pi.on('session_shutdown', async (_event, ctx) => {", + ' resetPromptState()', ' stopAnimation(ctx)', ' })', '}', diff --git a/src/main/project-groups/nested-repo-import.test.ts b/src/main/project-groups/nested-repo-import.test.ts index 2b591fa9a71..aa8a0502e1f 100644 --- a/src/main/project-groups/nested-repo-import.test.ts +++ b/src/main/project-groups/nested-repo-import.test.ts @@ -138,7 +138,9 @@ describe('createNestedProjectGroupResolver', () => { parentPath: '/workspace', groupName: 'workspace', mode: 'separate', - repoPaths: ['/workspace/services/api', '/workspace/services/worker'], + get repoPaths(): readonly string[] { + throw new Error('separate imports must not build unused folder scopes') + }, createGroup: () => { throw new Error('should not create a group') } @@ -148,6 +150,30 @@ describe('createNestedProjectGroupResolver', () => { expect(resolver.getCreatedGroups()).toEqual([]) }) + it('leaves every separate-import repo ungrouped even when repo paths are supplied', () => { + const { groups, createGroup } = createGroupRecorder() + const repoPaths = [ + '/workspace/services/api', + '/workspace/services/worker', + '/workspace/platform/packages/shared' + ] + const resolver = createNestedProjectGroupResolver({ + parentPath: '/workspace', + groupName: 'workspace', + mode: 'separate', + repoPaths, + createGroup + }) + + expect(repoPaths.map((repoPath) => resolver.getGroupForRepo(repoPath))).toEqual([ + undefined, + undefined, + undefined + ]) + expect(resolver.getRootGroup()).toBeUndefined() + expect(groups).toEqual([]) + }) + it('preserves filesystem root parent paths when creating the root group', () => { const groups: ProjectGroup[] = [] const resolver = createNestedProjectGroupResolver({ diff --git a/src/main/project-groups/nested-repo-import.ts b/src/main/project-groups/nested-repo-import.ts index 17aee4a2994..c08839edc30 100644 --- a/src/main/project-groups/nested-repo-import.ts +++ b/src/main/project-groups/nested-repo-import.ts @@ -150,10 +150,12 @@ export function createNestedProjectGroupResolver(args: { createGroup: (input: CreateGroupInput) => ProjectGroup }): NestedProjectGroupResolver { const createdGroups: ProjectGroup[] = [] - const folderScopes = buildSparseFolderScopes({ - parentPath: args.parentPath, - repoPaths: args.repoPaths ?? [] - }) + // Every folder-scope read sits behind ensureRootGroup, so outside group mode the scopes are + // unreachable. One flag drives both so the skip can never drift from the guard that justifies it. + const createsGroups = args.mode === 'group' + const folderScopes = createsGroups + ? buildSparseFolderScopes({ parentPath: args.parentPath, repoPaths: args.repoPaths ?? [] }) + : [] const folderScopesByRelativePath = new Map( folderScopes.map((scope) => [scope.relativePath, scope]) ) @@ -161,7 +163,7 @@ export function createNestedProjectGroupResolver(args: { let rootGroup: ProjectGroup | undefined const ensureRootGroup = (): ProjectGroup | undefined => { - if (args.mode !== 'group') { + if (!createsGroups) { return undefined } if (rootGroup) { diff --git a/src/main/runtime/mobile-session-terminal-retirement-proof.test.ts b/src/main/runtime/mobile-session-terminal-retirement-proof.test.ts index 8b5f4161a63..79aa4b27f75 100644 --- a/src/main/runtime/mobile-session-terminal-retirement-proof.test.ts +++ b/src/main/runtime/mobile-session-terminal-retirement-proof.test.ts @@ -1,7 +1,95 @@ import { describe, expect, it } from 'vitest' -import { appendRetiredTerminalSurfaceProofs } from './mobile-session-terminal-retirement-proof' +import { + appendRetiredTerminalSurfaceProofs, + preserveTerminalRetirementProofs +} from './mobile-session-terminal-retirement-proof' +import type { RuntimeMobileSessionTabsSnapshot } from '../../shared/runtime-types' + +const retired = { + parentTabId: 'tab', + leafId: 'leaf', + ptyId: 'pty', + terminal: 'term', + incarnationId: 'inc' +} +function snapshot( + overrides: Partial = {} +): RuntimeMobileSessionTabsSnapshot { + return { + worktree: 'worktree', + worktreeInstanceId: 'instance', + publicationEpoch: 'epoch', + snapshotVersion: 1, + activeGroupId: null, + activeTabId: null, + activeTabType: null, + tabs: [], + ...overrides + } +} describe('mobile session terminal retirement proofs', () => { + it.each([{ worktree: 'another-worktree' }, { worktreeInstanceId: 'successor-instance' }])( + 'does not copy retirement proof into a different workspace: %j', + (identity) => { + const next = snapshot(identity) + expect( + preserveTerminalRetirementProofs(next, snapshot({ retiredTerminalSurfaces: [retired] })) + ).toBe(next) + } + ) + + // Why: host-authored writes never carry worktreeInstanceId. Without inheritance the stored + // entry forgets the occupant and renderer(A) -> host write -> renderer(B) launders A's proofs. + it('does not launder proofs across occupants through an identity-less host write', () => { + const occupantA = snapshot({ + worktreeInstanceId: 'instance-a', + retiredTerminalSurfaces: [retired] + }) + const hostWrite = preserveTerminalRetirementProofs( + snapshot({ worktreeInstanceId: undefined, snapshotVersion: 2 }), + occupantA + ) + expect(hostWrite.worktreeInstanceId).toBe('instance-a') + expect(hostWrite.retiredTerminalSurfaces).toEqual([retired]) + + const occupantB = snapshot({ worktreeInstanceId: 'instance-b', snapshotVersion: 3 }) + expect(preserveTerminalRetirementProofs(occupantB, hostWrite)).toBe(occupantB) + }) + + it('keeps proofs for a host write that never learned any identity', () => { + const existing = snapshot({ worktreeInstanceId: undefined, retiredTerminalSurfaces: [retired] }) + const next = preserveTerminalRetirementProofs( + snapshot({ worktreeInstanceId: undefined, snapshotVersion: 2 }), + existing + ) + expect(next.worktreeInstanceId).toBeUndefined() + expect(next.retiredTerminalSurfaces).toEqual([retired]) + }) + + it('drops an old proof when its surface is published again', () => { + const existing = snapshot({ retiredTerminalSurfaces: [retired] }) + const revived = preserveTerminalRetirementProofs( + snapshot({ + tabs: [ + { + type: 'terminal', + id: 'tab::leaf', + parentTabId: 'tab', + leafId: 'leaf', + ptyId: 'successor-pty', + title: 'Successor', + isActive: false + } + ] + }), + existing + ) + expect(revived.retiredTerminalSurfaces).toEqual([]) + expect( + preserveTerminalRetirementProofs(snapshot(), revived).retiredTerminalSurfaces + ).toBeUndefined() + }) it('keeps the newest 64 exact identities', () => { let proofs = appendRetiredTerminalSurfaceProofs( undefined, diff --git a/src/main/runtime/mobile-session-terminal-retirement-proof.ts b/src/main/runtime/mobile-session-terminal-retirement-proof.ts index b1ce97e3f91..d5b6c620954 100644 --- a/src/main/runtime/mobile-session-terminal-retirement-proof.ts +++ b/src/main/runtime/mobile-session-terminal-retirement-proof.ts @@ -1,28 +1,50 @@ -import type { RuntimeMobileSessionRetiredTerminalSurface } from '../../shared/runtime-types' +import type { RuntimeMobileSessionTabsSnapshot } from '../../shared/runtime-types' +import { + appendRetiredTerminalSurfaceProofs, + dropRetirementProofsForLiveSurfaces +} from '../../shared/terminal-retirement-proof-ledger' -const MAX_RETIRED_TERMINAL_SURFACE_PROOFS = 64 +export { + appendRetiredTerminalSurfaceProofs, + dropRetirementProofsForLiveSurfaces +} from '../../shared/terminal-retirement-proof-ledger' -export function appendRetiredTerminalSurfaceProofs( - existing: readonly RuntimeMobileSessionRetiredTerminalSurface[] | undefined, - retired: readonly RuntimeMobileSessionRetiredTerminalSurface[] -): RuntimeMobileSessionRetiredTerminalSurface[] { - const next = new Map( - (existing ?? []).map((surface) => [ - `${surface.parentTabId}\0${surface.leafId}\0${surface.terminal}`, - surface - ]) - ) - for (const evidence of retired) { - const key = `${evidence.parentTabId}\0${evidence.leafId}\0${evidence.terminal}` - next.delete(key) - next.set(key, evidence) +/** + * Renderer snapshots omit the host's durable close acknowledgements; carry them forward. + * + * Why the identity inheritance: host-authored writes never set `worktreeInstanceId`. Without it + * the stored entry forgets which occupant minted the proofs, and renderer(A) -> host write -> + * renderer(B) would launder A's proofs into B. + */ +export function preserveTerminalRetirementProofs( + snapshot: RuntimeMobileSessionTabsSnapshot, + existing: RuntimeMobileSessionTabsSnapshot | undefined +): RuntimeMobileSessionTabsSnapshot { + if (!existing || existing.worktree !== snapshot.worktree) { + return snapshot } - while (next.size > MAX_RETIRED_TERMINAL_SURFACE_PROOFS) { - const oldest = next.keys().next().value - if (typeof oldest !== 'string') { - break - } - next.delete(oldest) + if ( + existing.worktreeInstanceId !== undefined && + snapshot.worktreeInstanceId !== undefined && + existing.worktreeInstanceId !== snapshot.worktreeInstanceId + ) { + return snapshot + } + const identified = + snapshot.worktreeInstanceId === undefined && existing.worktreeInstanceId !== undefined + ? { ...snapshot, worktreeInstanceId: existing.worktreeInstanceId } + : snapshot + if (!existing.retiredTerminalSurfaces?.length) { + return identified + } + return { + ...identified, + retiredTerminalSurfaces: dropRetirementProofsForLiveSurfaces( + appendRetiredTerminalSurfaceProofs( + existing.retiredTerminalSurfaces, + snapshot.retiredTerminalSurfaces ?? [] + ), + snapshot.tabs + ) } - return [...next.values()] } diff --git a/src/main/runtime/orca-runtime-fence-automation-owner.ts b/src/main/runtime/orca-runtime-automation-operations.ts similarity index 97% rename from src/main/runtime/orca-runtime-fence-automation-owner.ts rename to src/main/runtime/orca-runtime-automation-operations.ts index a90730c7886..fe65979ed1e 100644 --- a/src/main/runtime/orca-runtime-fence-automation-owner.ts +++ b/src/main/runtime/orca-runtime-automation-operations.ts @@ -23,7 +23,7 @@ import type { LegacyWorkerTerminalRecoveryResult } from './runtime-legacy-worker import { makePaneKey } from '../../shared/stable-pane-id' import { runtimeWorktreeIdsEqual } from './runtime-worktree-path-identity' -export class OrcaRuntimeWithFenceAutomationOwner extends OrcaRuntimeWithPtyForegroundProcessReads { +export class OrcaRuntimeWithAutomationOperations extends OrcaRuntimeWithPtyForegroundProcessReads { protected fenceAutomationOwner( id: string, expectedOwner: AutomationOwnerPrecondition | undefined, @@ -167,10 +167,6 @@ export class OrcaRuntimeWithFenceAutomationOwner extends OrcaRuntimeWithPtyForeg this.scheduleRestoredMessageRepoints() } - prepareLegacyWorkerTerminalRecovery(): LegacyWorkerTerminalRecoveryPlan { - return this.legacyWorkerRecovery.prepare() - } - protected async flushWorkspaceSessionOrThrowAsync(): Promise { const store = this.store if (store?.flushPendingOrThrowAsync) { diff --git a/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts b/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts index 31fb8a90ab0..1bb45f838e0 100644 --- a/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts +++ b/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts @@ -141,8 +141,10 @@ export class OrcaRuntimeWithCreateRuntimeOwnedMobileSessionTerminal extends Orca ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.storeMobileSessionSnapshot(worktreeId, next) - const result = this.toMobileSessionTabsResult(next) + // Why: emit the stored snapshot, not the pre-store one — storing grafts on retirement + // proofs, and subscribers dedupe on version so they would never see them otherwise. + const stored = this.storeMobileSessionSnapshot(worktreeId, next) + const result = this.toMobileSessionTabsResult(stored) const changeSequence = ++this.mobileSessionTabsChangeSequence for (const subscription of this.mobileSessionTabListeners) { subscription.listener( diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 240156d93c6..3dbd8809018 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -1,4 +1,5 @@ // @ts-nocheck -- mechanically split from OrcaRuntimeService; behavior is covered by AST equivalence and characterization tests. +import { collectRuntimeWorktreeAgentSources } from './runtime-worktree-agent-sources' import { OrcaRuntimeWithStructuredAgentSessionRecoverTuiOwner } from './orca-runtime-structured-agent-session-recover-tui-owner' import { DEFAULT_WORKTREE_PS_LIMIT } from './orca-runtime-postlude' import type { RuntimeWorktreePsResult } from '../../shared/runtime-types' @@ -9,6 +10,7 @@ import { applyRuntimeWorktreePsTerminalActivity } from './runtime-worktree-ps-activity' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' +import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import { compareWorktreePs } from './runtime-worktree-status-projection' import type { AgentSessionRecord } from '../../shared/agent-session-record' import type { Repo } from '../../shared/repo-types' @@ -102,11 +104,15 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent summaries, pathIndex: runtimeWorktreeSummaryPathIndex, missingWorktreeIds: missingRuntimeWorktreeIds, - mirroredWorktreeIdByTabId, - connectedPtyEvidence, workingTerminalEvidenceByWorktreeId, - retainedSnapshots: this.agentRows.values(), - hookSnapshots: this.getAgentStatusSnapshotFn?.() ?? [], + rowSources: collectRuntimeWorktreeAgentSources({ + mirroredWorktreeIdByTabId, + connectedPtyEvidence, + retainedSnapshots: this.agentRows.values(), + hookSnapshots: this.getAgentStatusSnapshotFn?.() ?? [], + // Broadcast history outlives closed sessions; only the host roster is eligible. + structuredSummaries: getStructuredAgentSessionHost()?.liveSessionStatusSummaries() ?? [] + }), orchestrationByPaneKey: this.agentOrchestrationProjection.buildByPaneKey(), getSummary: (summaryMap, pathIndex, missingIds, worktreeId) => this.getSummaryForRuntimeWorktreeId(summaryMap, pathIndex, missingIds, worktreeId) diff --git a/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts b/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts index 6956644c748..d0425cdc1f9 100644 --- a/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts +++ b/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts @@ -1,5 +1,5 @@ // @ts-nocheck -- mechanically split from OrcaRuntimeService; behavior is covered by AST equivalence and characterization tests. -import { OrcaRuntimeWithFenceAutomationOwner } from './orca-runtime-fence-automation-owner' +import { OrcaRuntimeWithAutomationOperations } from './orca-runtime-automation-operations' import { resolveTerminalSessionWorktreeId, runtimeWorktreeIdsEqual @@ -26,7 +26,7 @@ import type { ArtifactWriteRequest } from '../../shared/artifacts' -export class OrcaRuntimeWithHasExactPersistedTerminalSurfaceIdentity extends OrcaRuntimeWithFenceAutomationOwner { +export class OrcaRuntimeWithHasExactPersistedTerminalSurfaceIdentity extends OrcaRuntimeWithAutomationOperations { protected hasExactPersistedTerminalSurfaceIdentity(expected: { worktreeId: string tabId: string diff --git a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts index d53994032a1..2e1357e3450 100644 --- a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts +++ b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts @@ -136,8 +136,7 @@ export class OrcaRuntimeWithPreservedBranchCleanup extends OrcaRuntimeWithTermin new RuntimeLegacyWorkerTerminalRecoveryPersistence( () => this.store, () => this.getOrchestrationDb(), - (worktreeId) => this.tryGetWorkspaceSessionHostIdForWorktree(worktreeId), - (paneKey, blocked) => this.notifier?.setLegacyWorkerTerminalResumeFence?.(paneKey, blocked) + (worktreeId) => this.tryGetWorkspaceSessionHostIdForWorktree(worktreeId) ) protected readonly legacyWorkerRecovery = new RuntimeLegacyWorkerTerminalRecoveryController({ diff --git a/src/main/runtime/orca-runtime-register-pty.ts b/src/main/runtime/orca-runtime-register-pty.ts index 410798a329d..dae2519ca9f 100644 --- a/src/main/runtime/orca-runtime-register-pty.ts +++ b/src/main/runtime/orca-runtime-register-pty.ts @@ -80,6 +80,15 @@ export class OrcaRuntimeWithRegisterPty extends OrcaRuntimeWithInvalidateAllHand ...(binding && paneKey ? { tabId: binding.tabId, paneKey } : {}), ...(binding?.incarnationId ? { incarnationId: binding.incarnationId } : {}) }) + const hostScope = this.getOrchestrationCompatibilityHostScope(pty) + if (paneKey && binding?.incarnationId && hostScope) { + this._orchestrationDb?.retainReplacedWorkerTerminalResources({ + paneKey, + worktreeId, + hostScope: JSON.stringify(hostScope), + processIncarnation: `${ptyId}:${binding.incarnationId}` + }) + } const agentLaunchAuthority = binding?.agentLaunchAuthority if ( agentLaunchAuthority && diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index e912cc6b665..f138970d5ad 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -124,9 +124,9 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu ), tabs: existing.tabs.map((tab) => ({ ...tab, isActive: tab.id === id })) } - this.storeMobileSessionSnapshot(input.workspaceId, snapshot) + const stored = this.storeMobileSessionSnapshot(input.workspaceId, snapshot) if (input.notify !== false) { - this.emitMobileSessionTabsSnapshot(snapshot) + this.emitMobileSessionTabsSnapshot(stored) } return } @@ -174,9 +174,9 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.storeMobileSessionSnapshot(input.workspaceId, snapshot) + const stored = this.storeMobileSessionSnapshot(input.workspaceId, snapshot) if (input.notify !== false) { - this.emitMobileSessionTabsSnapshot(snapshot) + this.emitMobileSessionTabsSnapshot(stored) } } diff --git a/src/main/runtime/orca-runtime-runtime-id.ts b/src/main/runtime/orca-runtime-runtime-id.ts index 87dda7ebde0..caf23eee00b 100644 --- a/src/main/runtime/orca-runtime-runtime-id.ts +++ b/src/main/runtime/orca-runtime-runtime-id.ts @@ -1,5 +1,6 @@ // @ts-nocheck -- mechanically split from OrcaRuntimeService; behavior is covered by AST equivalence and characterization tests. import { randomUUID } from 'node:crypto' +import { preserveTerminalRetirementProofs } from './mobile-session-terminal-retirement-proof' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import { replaceConversationInSnapshot } from './structured-conversation-tab-replacement' import type { RuntimeStore } from './runtime-store-contract' @@ -107,6 +108,7 @@ export class OrcaRuntimeWithRuntimeId { snapshot = replaceConversationInSnapshot(snapshot, replacement) } const existing = this.mobileSessionTabsByWorktree.get(worktreeId) + snapshot = preserveTerminalRetirementProofs(snapshot, existing) const snapshotVersion = existing ? Math.max(snapshot.snapshotVersion, existing.snapshotVersion + 1) : snapshot.snapshotVersion diff --git a/src/main/runtime/orca-runtime-serialize-headless-terminal-buffer.ts b/src/main/runtime/orca-runtime-serialize-headless-terminal-buffer.ts index 8f4ddc430c5..673c31167f4 100644 --- a/src/main/runtime/orca-runtime-serialize-headless-terminal-buffer.ts +++ b/src/main/runtime/orca-runtime-serialize-headless-terminal-buffer.ts @@ -109,6 +109,9 @@ export class OrcaRuntimeWithSerializeHeadlessTerminalBuffer extends OrcaRuntimeW // still awaiting their first PTY (ptyId null) may adopt it, which preserves // the mobile pre-spawn subscribe flow. resolveLiveLeafForHandle(handle: string): { ptyId: string | null } | null { + // Why the discarded call: it re-links a runtime-owned handle whose `handles` record a renderer + // reload cleared, so the lookup below sees it; without it a phone's held handle inspects nothing. + this.getLivePtyForHandle(handle) const record = this.handles.get(handle) if (!record) { return null diff --git a/src/main/runtime/orca-runtime-subscribe-to-terminal-resize.ts b/src/main/runtime/orca-runtime-subscribe-to-terminal-resize.ts index d610dbb235f..1cef2bbf23a 100644 --- a/src/main/runtime/orca-runtime-subscribe-to-terminal-resize.ts +++ b/src/main/runtime/orca-runtime-subscribe-to-terminal-resize.ts @@ -54,16 +54,6 @@ export class OrcaRuntimeWithSubscribeToTerminalResize extends OrcaRuntimeWithApp // dispatch contexts immediately, rather than waiting for the coordinator's // next poll cycle. This catches agent crashes and unexpected exits within // milliseconds. The task is set back to 'pending' so it can be re-dispatched. - /** A worker settled by its own process exit makes its pane fenceable now, not at the next app - * start; a fence sweep must never fail the exit path behind it. */ - private sweepSettledWorkerResumeFencesAfterExit(): void { - try { - this.prepareLegacyWorkerTerminalRecovery() - } catch (error) { - console.warn('[orchestration] settled worker resume fence sweep failed', error) - } - } - protected failActiveDispatchOnExit( handle: string, paneKey: string | null, @@ -90,7 +80,6 @@ export class OrcaRuntimeWithSubscribeToTerminalResize extends OrcaRuntimeWithApp const stopping = this._orchestrationDb.getWorkerDispatch?.(dispatch.id) if (stopping?.state === 'stopping' && stopping.runtime_epoch === this.getRuntimeId()) { this._orchestrationDb.settleWorkerStop(dispatch.id) - this.sweepSettledWorkerResumeFencesAfterExit() return } @@ -99,7 +88,6 @@ export class OrcaRuntimeWithSubscribeToTerminalResize extends OrcaRuntimeWithApp workerProcessExited: true, terminationReason: cause.kind }) - this.sweepSettledWorkerResumeFencesAfterExit() if (isDeliberateTerminalExit(cause)) { return } @@ -145,7 +133,7 @@ export class OrcaRuntimeWithSubscribeToTerminalResize extends OrcaRuntimeWithApp exitCause: cause, handle }), - ...(recipient.runId ? { runId: recipient.runId } : {}) + runId: dispatch.run_id }) this.notifyMessageArrived(escalation.to_handle, escalation.type) } catch (error) { diff --git a/src/main/runtime/orca-runtime-terminal-handle-incarnation.test.ts b/src/main/runtime/orca-runtime-terminal-handle-incarnation.test.ts index 19605aa4ffa..9765465fdf0 100644 --- a/src/main/runtime/orca-runtime-terminal-handle-incarnation.test.ts +++ b/src/main/runtime/orca-runtime-terminal-handle-incarnation.test.ts @@ -53,6 +53,28 @@ function register(runtime: OrcaRuntimeService, incarnationId: string): void { } describe('runtime terminal handle incarnation fencing', () => { + it('inspects the retained SSH PTY during renderer reload without an input write', async () => { + const { runtime } = makeRuntime() + const handle = runtime.preAllocateHandleForPty(PTY_ID) + register(runtime, 'incarnation-1') + syncGraph(runtime) + const inspectProcess = vi.fn().mockResolvedValue({ foregroundProcess: 'codex' }) + runtime.setPtyController({ + write: vi.fn(() => true), + kill: () => true, + getForegroundProcess: async () => null, + inspectProcess + }) + expect(runtime.markRendererReloading(1)).not.toBeNull() + expect((runtime as unknown as { handles: Map }).handles.has(handle)).toBe( + false + ) + await expect( + runtime.inspectTerminalProcess(handle, { expectedIncarnationId: 'incarnation-1' }) + ).resolves.toEqual({ foregroundProcess: 'codex' }) + expect(inspectProcess).toHaveBeenCalledWith(PTY_ID, { expectedIncarnationId: 'incarnation-1' }) + }) + it('preserves a direct handle while the PTY incarnation is unchanged', async () => { const { runtime } = makeRuntime() const handle = runtime.preAllocateHandleForPty(PTY_ID) diff --git a/src/main/runtime/orca-runtime-tests/lineage-and-scan-cache-part-05.spec.ts b/src/main/runtime/orca-runtime-tests/lineage-and-scan-cache-part-05.spec.ts index 1df978ac165..ad66e89ec7e 100644 --- a/src/main/runtime/orca-runtime-tests/lineage-and-scan-cache-part-05.spec.ts +++ b/src/main/runtime/orca-runtime-tests/lineage-and-scan-cache-part-05.spec.ts @@ -95,7 +95,10 @@ describe('OrcaRuntimeService', () => { return [name, createRootDispatch(db, task.id, handles[name], paneKey(name))] }) ) - const legacyTask = db.createTask({ spec: 'legacy worker' }) + const legacyTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'legacy worker' + }) const legacyDispatch = createRootDispatch( db, legacyTask.id, diff --git a/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-02.spec.ts b/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-02.spec.ts index 57c1d2c3031..9291c30c1a2 100644 --- a/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-02.spec.ts +++ b/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-02.spec.ts @@ -216,7 +216,10 @@ describe('OrcaRuntimeService', () => { runtime as unknown as { leaves: Map< string, - { lastAgentStatus: string | null; lastAgentStatusObservedLive: boolean } + { + lastAgentStatus: string | null + lastAgentStatusObservedLive: boolean + } > } ).leaves.values() @@ -415,7 +418,12 @@ describe('OrcaRuntimeService', () => { const [terminal] = (await runtime.listTerminals()).terminals runtime.onPtyData('pty-1', '\x1b]0;Codex working\x07', 100) - db.insertMessage({ from: 'term_worker', to: terminal.handle, subject: 'pending' }) + db.insertMessage({ + runId: 'run_legacy_local', + from: 'term_worker', + to: terminal.handle, + subject: 'pending' + }) runtime.notifyMessageArrived(terminal.handle, 'status') db.close() diff --git a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts index 3c61985e597..a2d45395b26 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts @@ -390,7 +390,7 @@ describe('OrcaRuntimeService', () => { expect(getSession().terminalTopologyRevisionByRepoId?.[TEST_REPO_ID]).toBe(1) }) - it('fences provider resume and reveals one exact live legacy worker without stealing focus', async () => { + it('reveals one exact live legacy worker without stealing focus', async () => { const workerLeafId = HEADLESS_LEAF_ID const coordinatorLeafId = HEADLESS_SECOND_LEAF_ID const workerPaneKey = `legacy-worker:${workerLeafId}` @@ -526,10 +526,7 @@ describe('OrcaRuntimeService', () => { resolveLegacyWorkerTerminalRecovery } as never) - runtime.prepareLegacyWorkerTerminalRecovery() - expect( - getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() const recovered = await runtime.reconcileLegacyWorkerTerminals({ materializeRenderer: true diff --git a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-03.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-03.spec.ts index 44edd9aa371..00a08310877 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-03.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-03.spec.ts @@ -17,7 +17,7 @@ import { } from '../orca-runtime-test-scenario-builders.spec' describe('OrcaRuntimeService', () => { - it('retries renderer reveal before clearing an adopted legacy worker resume fence', async () => { + it('retries renderer reveal before clearing an adopted legacy worker sleeping record', async () => { const workerPaneKey = `legacy-worker:${HEADLESS_LEAF_ID}` const incarnationId = '44444444-4444-4444-8444-444444444444' const session: WorkspaceSessionState = { @@ -106,7 +106,7 @@ describe('OrcaRuntimeService', () => { expect(resolveLegacyWorkerTerminalRecovery).toHaveBeenCalledWith(workerPaneKey, 'adopted') }) - it('keeps a revealed worker fenced until its exact renderer graph is published', async () => { + it('defers a revealed worker until its exact renderer graph is published', async () => { vi.useFakeTimers() try { const harness = makePostRevealWorkerRecoveryHarness(() => true) @@ -288,7 +288,7 @@ describe('OrcaRuntimeService', () => { } }) - it('keeps recovery fenced when the renderer omits the exact reveal identity', async () => { + it('defers recovery when the renderer omits the exact reveal identity', async () => { const harness = makePostRevealWorkerRecoveryHarness(() => false) harness.revealTerminalSession.mockResolvedValue({ tabId: 'legacy-post-reveal' }) @@ -417,7 +417,7 @@ describe('OrcaRuntimeService', () => { ) }) - it('keeps the legacy worker resume fence in memory when persistence fails', async () => { + it('keeps the legacy worker sleeping record in memory when persistence fails', async () => { const workerPaneKey = `legacy-worker:${HEADLESS_LEAF_ID}` const incarnationId = '99999999-9999-4999-8999-999999999999' const session: WorkspaceSessionState = { @@ -526,9 +526,7 @@ describe('OrcaRuntimeService', () => { }) expect(flushPendingOrThrowAsync).toHaveBeenCalledTimes(2) expect(revealTerminalSession).toHaveBeenCalledOnce() - expect( - getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() expect(getSession().sleepingAgentSessionsByPaneKey?.[concurrentPaneKey]?.tabId).toBe( 'concurrent-tab' ) diff --git a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-04.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-04.spec.ts index d09b4c64ce8..ea70b730388 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-04.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-04.spec.ts @@ -56,7 +56,10 @@ describe('OrcaRuntimeService', () => { ) const db = new OrchestrationDb(':memory:') try { - const task = db.createTask({ spec: 'continue after missing worker recovery' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'continue after missing worker recovery' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -163,7 +166,10 @@ describe('OrcaRuntimeService', () => { ) const db = new OrchestrationDb(':memory:') try { - const task = db.createTask({ spec: 'retry missing worker recovery' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'retry missing worker recovery' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -328,10 +334,7 @@ describe('OrcaRuntimeService', () => { resolveLegacyWorkerTerminalRecovery } as never) - runtime.prepareLegacyWorkerTerminalRecovery() - expect( - getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() await expect(runtime.reconcileLegacyWorkerTerminals()).resolves.toMatchObject({ adoptedDispatchIds: ['dispatch-exited-two'], @@ -445,9 +448,7 @@ describe('OrcaRuntimeService', () => { exitedDispatchIds: [], deferredDispatchIds: ['dispatch-inventory-unavailable'] }) - expect( - getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() expect(resolveLegacyWorkerTerminalRecovery).not.toHaveBeenCalled() expect(listProcesses).toHaveBeenCalledOnce() expect(getSession().tabsByWorktree[TEST_WORKTREE_ID]).toEqual([]) @@ -477,7 +478,6 @@ describe('OrcaRuntimeService', () => { try { const runtime = new OrcaRuntimeService(store) const reconcile = vi.spyOn(runtime, 'reconcileLegacyWorkerTerminals').mockResolvedValue({ - blockedPaneCount: 1, adoptedDispatchIds: [], exitedDispatchIds: [], deferredDispatchIds: [] diff --git a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-05.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-05.spec.ts index 5798916570e..389c70d4f42 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-05.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-05.spec.ts @@ -18,13 +18,12 @@ import { TEST_WORKTREE_PATH, makeFolderProjectGroup, makeFolderWorkspace, - makeRuntimeStoreWithWorkspaceSession, - store + makeRuntimeStoreWithWorkspaceSession } from '../orca-runtime-test-fixtures.spec' import { publishLegacyWorkerReveal } from '../orca-runtime-test-scenario-builders.spec' describe('OrcaRuntimeService', () => { - it('keeps live workers fenced without exact controller identity evidence', async () => { + it('defers live workers without exact controller identity evidence', async () => { const incarnationId = '56565656-5656-4656-8656-565656565656' const cases = [ { @@ -152,8 +151,7 @@ describe('OrcaRuntimeService', () => { for (const { name, leafId } of cases.slice(0, 2)) { expect( getSession().sleepingAgentSessionsByPaneKey?.[`legacy-${name}:${leafId}`] - ?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + ).toBeDefined() } for (const { name, leafId } of cases.slice(2)) { expect( @@ -374,12 +372,7 @@ describe('OrcaRuntimeService', () => { } as never) try { - expect(runtime.prepareLegacyWorkerTerminalRecovery()).toMatchObject({ - blockedPanes: [expect.objectContaining({ paneKey: workerPaneKey })] - }) - expect( - sshSession.sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(sshSession.sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() expect(localSession.sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeUndefined() await expect( runtime.reconcileLegacyWorkerTerminals({ @@ -422,74 +415,4 @@ describe('OrcaRuntimeService', () => { } }) }) - - it('fences an unresolved folder legacy worker in its exact retained session partition', () => { - const connectionId = 'ssh-unresolved-folder' - const worktreeId = 'folder:missing-folder' - const workerPaneKey = `legacy-unresolved-folder-worker:${HEADLESS_LEAF_ID}` - const remoteInitialSession: WorkspaceSessionState = { - ...getDefaultWorkspaceSession(), - tabsByWorktree: { [worktreeId]: [] }, - sleepingAgentSessionsByPaneKey: { - [workerPaneKey]: { - paneKey: workerPaneKey, - tabId: 'legacy-unresolved-folder-worker', - worktreeId, - agent: 'codex', - providerSession: { key: 'session_id', id: 'legacy-unresolved-folder-session' }, - prompt: 'continue', - state: 'working', - capturedAt: 1, - updatedAt: 1, - origin: 'live', - connectionId - } - } - } - const localSession = getDefaultWorkspaceSession() - let remoteSession = remoteInitialSession - const getWorkspaceSession = vi.fn((hostId?: string | null) => - hostId === `ssh:${connectionId}` ? remoteSession : localSession - ) - const setWorkspaceSession = vi.fn((next: WorkspaceSessionState, hostId?: string | null) => { - if (hostId !== `ssh:${connectionId}`) { - throw new Error(`unexpected workspace-session host ${hostId ?? 'default'}`) - } - remoteSession = next - }) - const runtime = new OrcaRuntimeService({ - ...store, - getFolderWorkspaces: () => [], - getWorkspaceSession, - getWorkspaceSessionHostIds: () => ['local', `ssh:${connectionId}`], - setWorkspaceSession, - flushOrThrow: vi.fn() - } as never) - runtime.setOrchestrationDb({ - listLegacyWorkerTerminalRecoveryRows: () => [ - { - dispatch_id: 'dispatch-unresolved-folder', - task_id: 'task-unresolved-folder', - dispatch_status: 'completed', - contract_version: 0, - assignee_handle: 'term_unresolved_folder', - assignee_pane_key: workerPaneKey, - process_incarnation: 'pty-unresolved-folder:68686868-6868-4868-8868-686868686868', - worker_state: 'ready', - worktree_id: worktreeId, - agent_terminal_handle: 'term_unresolved_folder' - } - ] - } as unknown as OrchestrationDb) - - expect(runtime.prepareLegacyWorkerTerminalRecovery()).toMatchObject({ - blockedPanes: [expect.objectContaining({ paneKey: workerPaneKey, worktreeId })] - }) - expect( - remoteSession.sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') - expect(localSession.sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeUndefined() - expect(setWorkspaceSession).toHaveBeenCalledOnce() - expect(setWorkspaceSession).toHaveBeenCalledWith(expect.any(Object), `ssh:${connectionId}`) - }) }) diff --git a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-06.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-06.spec.ts index 4078a291ab5..1907b2bb450 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-06.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-06.spec.ts @@ -153,11 +153,8 @@ describe('OrcaRuntimeService', () => { deferredDispatchIds: ['dispatch-ssh'] }) expect(listProcesses).not.toHaveBeenCalled() - expect( - getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() expect(localSession.sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeUndefined() - expect(getWorkspaceSession).toHaveBeenCalledWith(`ssh:${connectionId}`) await expect( runtime.reconcileLegacyWorkerTerminals({ @@ -175,6 +172,7 @@ describe('OrcaRuntimeService', () => { } expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeUndefined() + expect(getWorkspaceSession).toHaveBeenCalledWith(`ssh:${connectionId}`) expect(setWorkspaceSession).toHaveBeenCalledWith(expect.any(Object), `ssh:${connectionId}`) expect(listProcesses).toHaveBeenCalledTimes(3) expect(revealTerminalSession).toHaveBeenCalledWith(TEST_WORKTREE_ID, { @@ -297,9 +295,7 @@ describe('OrcaRuntimeService', () => { exitedDispatchIds: [], deferredDispatchIds: ['dispatch-wsl'] }) - expect( - getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]?.automaticResumeBlockedBy - ).toBe('legacy-orchestration-worker') + expect(getSession().sleepingAgentSessionsByPaneKey?.[workerPaneKey]).toBeDefined() expect(revealTerminalSession).not.toHaveBeenCalled() observedDistro = 'Ubuntu' diff --git a/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts new file mode 100644 index 00000000000..58e7de1b674 --- /dev/null +++ b/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts @@ -0,0 +1,146 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from '../orca-runtime-test-mocks.spec' +import { TEST_WORKTREE_ID, store } from '../orca-runtime-test-fixtures.spec' +import type { AgentJournalRenderItem } from '../../../shared/agent-session-journal-types' +import type { + AgentSessionStatusEvent, + AgentSessionStatusSummary +} from '../../../shared/agent-session-wire' +import type { AgentSessionJournal } from '../../native-chat/agent-session-journal/journal-store' +import type { StructuredAgentSessionHost } from '../../native-chat/agent-session-wire/structured-agent-session-host' +import { + getStructuredAgentSessionHost, + setStructuredAgentSessionHost +} from '../../native-chat/agent-session-wire/structured-agent-session-registry' +import { StructuredAgentSessionStatusFeed } from '../../native-chat/agent-session-wire/structured-agent-session-status-feed' + +/** + * The production wiring, not the projection. Both structured-row suites call + * `attachRuntimeWorktreeAgentRows` directly with summaries they built themselves, so nothing + * executed `getWorktreePs`'s own `getStructuredAgentSessionHost()?.liveSessionStatusSummaries()` + * — and that file carries `@ts-nocheck`, so renaming the accessor stayed green in typecheck AND + * in the suite while `orca worktree ps` and mobile's poll would throw for every user. + */ + +const HELD_SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' +const FORGOTTEN_SESSION = 'b2c3d4e5-f6a7-4b8c-9d0e-1f2a3b4c5d6e' +const OBSERVED_AT = 1_757_030_400_000 + +function runningTurn(prompt: string): AgentJournalRenderItem[] { + return [ + { + itemId: 'user-1', + sequence: 1, + revision: 1, + observedAt: OBSERVED_AT, + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: prompt }] } + }, + { + itemId: 'turn-1', + sequence: 2, + revision: 1, + observedAt: OBSERVED_AT, + body: { + kind: 'status', + text: 'Working', + turnLifecycle: { turnId: 'turn-1', state: 'running' } + } + } + ] +} + +function journalWith(prompt: string): AgentSessionJournal { + return { + isReadOnly: false, + lastActivityAt: () => OBSERVED_AT, + snapshot: () => ({ items: runningTurn(prompt) }) + } as unknown as AgentSessionJournal +} + +/** A real feed the host still holds one session on, having forgotten the other. Its `published` + * cache never retracts, so the two views genuinely differ. */ +function statusFeed(): StructuredAgentSessionStatusFeed { + const session = (prompt: string) => ({ + journal: journalWith(prompt), + params: { location: { workspaceId: TEST_WORKTREE_ID }, provider: 'claude' as const } + }) + const sessions = new Map([ + [HELD_SESSION, session('ship the thing')], + [FORGOTTEN_SESSION, session('rm the branch')] + ]) + const feed = new StructuredAgentSessionStatusFeed({ + sessions, + getRecord: () => null, + now: () => OBSERVED_AT + }) + feed.publish(HELD_SESSION) + feed.publish(FORGOTTEN_SESSION) + // `forget-session`, the last eviction step, drops the session and leaves the feed alone. + sessions.delete(FORGOTTEN_SESSION) + return feed +} + +/** Everything the feed retained, read through the snapshot a subscriber opens on. No production + * code reads this; it is here so swapping the call site back to a whole-cache read is a one-line + * edit that this suite must catch. */ +function retainedSummaries(feed: StructuredAgentSessionStatusFeed): AgentSessionStatusSummary[] { + let retained: AgentSessionStatusSummary[] = [] + feed.subscribe({ + id: 'retained-probe', + emit: (event: AgentSessionStatusEvent) => { + if (event.type === 'snapshot') { + retained = event.sessions + } + } + })() + return retained +} + +function installHost(feed: StructuredAgentSessionStatusFeed) { + const liveSessionStatusSummaries = vi.fn(() => feed.liveSessionSummaries()) + const retainedSessionStatusSummaries = vi.fn(() => retainedSummaries(feed)) + // Typed against the real host, so renaming the accessor on the class reddens `tc` here — the + // caller cannot, because `orca-runtime-get-worktree-ps.ts` is `@ts-nocheck`. + const host: Pick & { + retainedSessionStatusSummaries: () => AgentSessionStatusSummary[] + } = { liveSessionStatusSummaries, retainedSessionStatusSummaries } + setStructuredAgentSessionHost(host as unknown as StructuredAgentSessionHost) + return { liveSessionStatusSummaries, retainedSessionStatusSummaries } +} + +describe('worktree ps reads the installed structured host', () => { + afterEach(() => { + setStructuredAgentSessionHost(null) + }) + + it('reports the held session and asks the host for its live summaries', async () => { + const feed = statusFeed() + const { liveSessionStatusSummaries } = installHost(feed) + + const { worktrees } = await new OrcaRuntimeService(store).getWorktreePs() + + const worktree = worktrees.find((entry) => entry.worktreeId === TEST_WORKTREE_ID) + expect(worktree).toBeDefined() + // Exactly one: the forgotten session is still in the feed's retained cache, so a call site + // that enumerated that cache instead would report two. + expect(worktree?.agents).toHaveLength(1) + expect(worktree?.agents[0]).toMatchObject({ + state: 'working', + agentType: 'claude', + prompt: 'ship the thing' + }) + // Pins the call site to the live-intersecting accessor, not merely to some accessor. + expect(liveSessionStatusSummaries).toHaveBeenCalledTimes(1) + }) + + it('succeeds with no structured rows when no host is installed', async () => { + // Guard the guard: these specs share one module registry, so state the premise. + expect(getStructuredAgentSessionHost()).toBeNull() + + const { worktrees } = await new OrcaRuntimeService(store).getWorktreePs() + + const worktree = worktrees.find((entry) => entry.worktreeId === TEST_WORKTREE_ID) + expect(worktree).toBeDefined() + expect(worktree?.agents).toEqual([]) + }) +}) diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index 829c2ca321e..74098040c27 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -85,6 +85,7 @@ await import('./orca-runtime-tests/mobile-summaries.spec') await import('./orca-runtime-tests/mobile-summaries-part-02.spec') await import('./orca-runtime-tests/mobile-summaries-part-03.spec') await import('./orca-runtime-tests/mobile-summaries-part-04.spec') +await import('./orca-runtime-tests/worktree-ps-structured-host.spec') await import('./orca-runtime-tests/terminal-sleep-and-teardown.spec') await import('./orca-runtime-tests/terminal-sleep-and-teardown-part-02.spec') await import('./orca-runtime-tests/terminal-sleep-and-teardown-part-03.spec') diff --git a/src/main/runtime/orchestration-messages-fake-parity.test.ts b/src/main/runtime/orchestration-messages-fake-parity.test.ts index 72ed8695b5b..5a785fc8602 100644 --- a/src/main/runtime/orchestration-messages-fake-parity.test.ts +++ b/src/main/runtime/orchestration-messages-fake-parity.test.ts @@ -7,9 +7,13 @@ type PointerTarget = { ptyId: string; processIncarnation: string } // The slice of the mailbox store the pointer batch selector depends on. type PointerStore = { - insertMessage(message: { from: string; to: string; subject: string; type?: MessageType }): { - id: string - } + insertMessage(message: { + runId: string + from: string + to: string + subject: string + type?: MessageType + }): { id: string } stageMailboxPointerEnter(ids: string[], target: PointerTarget): boolean markMailboxPointerWriteAttempted(ids: string[], target: PointerTarget): boolean getUndeliveredUnreadMessages( @@ -32,7 +36,12 @@ describe.each(STORES)('mailbox pointer reservations (%s)', (_name, createStore) it('refuses a claim another flight already holds', () => { const store = createStore() - const message = store.insertMessage({ from: 'a', to: 'run:run-1', subject: 'contended' }) + const message = store.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'contended' + }) expect(store.stageMailboxPointerEnter([message.id], rival)).toBe(true) expect(store.stageMailboxPointerEnter([message.id], mine)).toBe(false) @@ -41,8 +50,18 @@ describe.each(STORES)('mailbox pointer reservations (%s)', (_name, createStore) it('rolls the whole batch back when one row is already claimed', () => { const store = createStore() - const free = store.insertMessage({ from: 'a', to: 'run:run-1', subject: 'free' }) - const taken = store.insertMessage({ from: 'a', to: 'run:run-1', subject: 'taken' }) + const free = store.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'free' + }) + const taken = store.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'taken' + }) expect(store.stageMailboxPointerEnter([taken.id], rival)).toBe(true) expect(store.stageMailboxPointerEnter([free.id, taken.id], mine)).toBe(false) @@ -52,8 +71,19 @@ describe.each(STORES)('mailbox pointer reservations (%s)', (_name, createStore) it('applies the exclusion and limit the pointer batch selector relies on', () => { const store = createStore() - store.insertMessage({ from: 'a', to: 'run:run-1', subject: 'reserved', type: 'escalation' }) - const kept = store.insertMessage({ from: 'a', to: 'run:run-1', subject: 'kept' }) + store.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'reserved', + type: 'escalation' + }) + const kept = store.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'kept' + }) expect( store diff --git a/src/main/runtime/orchestration/coordinator-decision-gates.test.ts b/src/main/runtime/orchestration/coordinator-decision-gates.test.ts index 15e0cd2c7b0..3e9934b85ad 100644 --- a/src/main/runtime/orchestration/coordinator-decision-gates.test.ts +++ b/src/main/runtime/orchestration/coordinator-decision-gates.test.ts @@ -12,13 +12,14 @@ describe('coordinator decision-gate authority', () => { it('opens a gate only for the sender-owned active Dispatch', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'owned gate target' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'owned gate target' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', 'tab_owner:leaf_owner') const logs: string[] = [] openDecisionGateFromMessage( db, db.insertMessage({ + runId: 'run_legacy_local', from: 'term_owner', to: 'term_coordinator', subject: 'Need approval', @@ -41,20 +42,27 @@ describe('coordinator decision-gate authority', () => { it('rejects a gate targeting another active Dispatch without mutating either Task', () => { db = new OrchestrationDb(':memory:') - const attackerTask = db.createTask({ spec: 'attacker assignment' }) + const attackerTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'attacker assignment' + }) const attacker = createRootDispatch( db, attackerTask.id, 'term_attacker', 'tab_attacker:leaf_attacker' ) - const victimTask = db.createTask({ spec: 'victim assignment' }) + const victimTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'victim assignment' + }) const victim = createRootDispatch(db, victimTask.id, 'term_victim', 'tab_victim:leaf_victim') const logs: string[] = [] openDecisionGateFromMessage( db, db.insertMessage({ + runId: 'run_legacy_local', from: 'term_attacker', to: 'term_coordinator', subject: 'Block the victim', @@ -79,12 +87,16 @@ describe('coordinator decision-gate authority', () => { it('accepts the canonical sender of an imported federated Dispatch', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'remote gate target' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'remote gate target' + }) const dispatch = createRootDispatch(db, task.id, 'remote-worker') openDecisionGateFromMessage( db, db.insertMessage({ + runId: 'run_legacy_local', from: `dispatch:${dispatch.id}`, to: 'term_coordinator', subject: 'Remote approval required', diff --git a/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts b/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts index 5f99e283de3..f0960803988 100644 --- a/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts +++ b/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts @@ -66,7 +66,7 @@ describe('coordinator dispatch with an unobserved prompt', () => { it('never re-pastes a preamble whose turn start was not observed', async () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'do the work' }) const runtime = createRuntime(new Error('agent_prompt_stalled')) const logs: string[] = [] @@ -88,7 +88,7 @@ describe('coordinator dispatch with an unobserved prompt', () => { it('lets a late worker report settle a dispatch whose prompt was unobserved', async () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'do the work' }) await dispatch(createRuntime(new Error('agent_prompt_stalled')), task.id, []) const dispatchId = db.getDispatchContext(task.id)!.id const minted = db.mintDispatchCapability({ @@ -119,7 +119,7 @@ describe('coordinator dispatch with an unobserved prompt', () => { it('still fails the dispatch when the prompt was never delivered', async () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'do the work' }) const runtime = createRuntime(new Error('terminal_not_writable')) await expect(dispatch(runtime, task.id, [])).rejects.toThrow('terminal_not_writable') diff --git a/src/main/runtime/orchestration/coordinator-drift-probe-coalescing.test.ts b/src/main/runtime/orchestration/coordinator-drift-probe-coalescing.test.ts index f42713c0dc9..5367af466bf 100644 --- a/src/main/runtime/orchestration/coordinator-drift-probe-coalescing.test.ts +++ b/src/main/runtime/orchestration/coordinator-drift-probe-coalescing.test.ts @@ -44,8 +44,8 @@ describe('Coordinator drift probe coalescing', () => { : { base: 'origin/main', behind: 0, recentSubjects: [] } } } - const first = db.createTask({ spec: 'first task' }) - const second = db.createTask({ spec: 'second task' }) + const first = db.createTask({ runId: 'run_legacy_local', spec: 'first task' }) + const second = db.createTask({ runId: 'run_legacy_local', spec: 'second task' }) const coordinator = new Coordinator(db, runtime, { spec: 'go', coordinatorHandle: 'coord', @@ -65,6 +65,7 @@ describe('Coordinator drift probe coalescing', () => { throw new Error(`missing dispatch for ${task.id}`) } db.insertMessage({ + runId: 'run_legacy_local', from: dispatch.assignee_handle, to: 'coord', subject: 'Done', @@ -105,8 +106,14 @@ describe('Coordinator drift probe coalescing', () => { } } } - const refused = db.createTask({ spec: 'requires a current base' }) - const allowed = db.createTask({ spec: 'can use stale base\nallow-stale-base: true' }) + const refused = db.createTask({ + runId: 'run_legacy_local', + spec: 'requires a current base' + }) + const allowed = db.createTask({ + runId: 'run_legacy_local', + spec: 'can use stale base\nallow-stale-base: true' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', coordinatorHandle: 'coord', diff --git a/src/main/runtime/orchestration/coordinator-escalation-triage.test.ts b/src/main/runtime/orchestration/coordinator-escalation-triage.test.ts index c020e62dca0..e0bdf75fc98 100644 --- a/src/main/runtime/orchestration/coordinator-escalation-triage.test.ts +++ b/src/main/runtime/orchestration/coordinator-escalation-triage.test.ts @@ -12,20 +12,21 @@ describe('coordinator escalation authority', () => { it('rejects an escalation targeting another active Dispatch', () => { db = new OrchestrationDb(':memory:') - const attackerTask = db.createTask({ spec: 'attacker assignment' }) + const attackerTask = db.createTask({ runId: 'run_legacy_local', spec: 'attacker assignment' }) const attacker = createRootDispatch( db, attackerTask.id, 'term_attacker', 'tab_attacker:leaf_attacker' ) - const victimTask = db.createTask({ spec: 'victim assignment' }) + const victimTask = db.createTask({ runId: 'run_legacy_local', spec: 'victim assignment' }) const victim = createRootDispatch(db, victimTask.id, 'term_victim') const logs: string[] = [] applyEscalationToDispatch( db, db.insertMessage({ + runId: 'run_legacy_local', from: 'term_attacker', to: 'term_coordinator', subject: 'Fail the victim', @@ -43,12 +44,13 @@ describe('coordinator escalation authority', () => { it('accepts the canonical sender of an imported federated Dispatch', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'remote escalation target' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'remote escalation target' }) const dispatch = createRootDispatch(db, task.id, 'remote-worker') applyEscalationToDispatch( db, db.insertMessage({ + runId: 'run_legacy_local', from: `dispatch:${dispatch.id}`, to: 'term_coordinator', subject: 'Remote worker failed', diff --git a/src/main/runtime/orchestration/coordinator-stale-base-flag.test.ts b/src/main/runtime/orchestration/coordinator-stale-base-flag.test.ts new file mode 100644 index 00000000000..818d3f848dc --- /dev/null +++ b/src/main/runtime/orchestration/coordinator-stale-base-flag.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from 'vitest' +import { parseAllowStaleBaseFromSpec } from './coordinator-stale-base-flag' + +describe('parseAllowStaleBaseFromSpec', () => { + it('matches canonical form on its own line and strips it', () => { + const spec = `Do the work +allow-stale-base: true` + const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) + expect(allowStale).toBe(true) + expect(strippedSpec).toBe('Do the work\n') + expect(strippedSpec).not.toContain('allow-stale-base') + }) + + it('matches case-insensitively', () => { + const spec = `Do the work +Allow-Stale-Base: TRUE` + const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) + expect(allowStale).toBe(true) + expect(strippedSpec).not.toMatch(/[Aa]llow-[Ss]tale-[Bb]ase/) + }) + + it('does not match allow-stale-base: false', () => { + const spec = `Do the work +allow-stale-base: false` + const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) + expect(allowStale).toBe(false) + expect(strippedSpec).toBe(spec) + }) + + it('does not match allow-stale-base: truthy', () => { + const spec = `Do the work +allow-stale-base: truthy` + const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) + expect(allowStale).toBe(false) + expect(strippedSpec).toBe(spec) + }) + + it('does not match the flag embedded inside a sentence', () => { + const spec = 'we allow-stale-base: true sometimes' + const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) + expect(allowStale).toBe(false) + expect(strippedSpec).toBe(spec) + }) + + it('handles the flag as the last line with no trailing newline', () => { + const spec = 'line 1\nallow-stale-base: true' + const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) + expect(allowStale).toBe(true) + expect(strippedSpec).toBe('line 1\n') + expect(strippedSpec.endsWith('allow-stale-base: true')).toBe(false) + }) +}) diff --git a/src/main/runtime/orchestration/coordinator.test.ts b/src/main/runtime/orchestration/coordinator.test.ts index 38701a9d7dd..75ba01ffa27 100644 --- a/src/main/runtime/orchestration/coordinator.test.ts +++ b/src/main/runtime/orchestration/coordinator.test.ts @@ -3,12 +3,11 @@ import { OrchestrationDb } from './db' import { reconcileLifecycleMessage } from './lifecycle-reconciliation' import { Coordinator } from './coordinator' import type { CoordinatorRuntime } from './coordinator-runtime-contract' -import { - DISPATCH_STALE_THRESHOLD, - parseAllowStaleBaseFromSpec -} from './coordinator-stale-base-flag' +import { DISPATCH_STALE_THRESHOLD } from './coordinator-stale-base-flag' import { createRootDispatch } from './db/root-dispatch-test-fixture' +const runId = 'run_legacy_local' + type DriftResult = { base: string behind: number @@ -92,6 +91,7 @@ function insertWorkerDone( } const from = params.from ?? dispatch?.assignee_handle ?? 'term_unknown' db.insertMessage({ + runId, from, to: params.to ?? 'coord', subject: 'Done', @@ -131,7 +131,10 @@ describe('Coordinator', () => { runtime.cliCommand = 'orca-ide' runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] - const task = db.createTask({ spec: 'implement feature' }) + const task = db.createTask({ + runId, + spec: 'implement feature' + }) // Simulate worker_done arriving after dispatch const coordinator = new Coordinator(db, runtime, { @@ -166,7 +169,10 @@ describe('Coordinator', () => { getTerminalPaneKey: (handle: string) => (handle === 'term_a' ? 'tab_a:leaf_a' : null) }) - const task = db.createTask({ spec: 'implement feature' }) + const task = db.createTask({ + runId, + spec: 'implement feature' + }) const coordinator = new Coordinator(db, withPaneLookup, { spec: 'build it', coordinatorHandle: 'coord', @@ -198,7 +204,10 @@ describe('Coordinator', () => { } : null }) - const task = db.createTask({ spec: 'implement feature' }) + const task = db.createTask({ + runId, + spec: 'implement feature' + }) const coordinator = new Coordinator(db, withAuthority, { spec: 'build it', coordinatorHandle: 'coord', @@ -223,9 +232,13 @@ describe('Coordinator', () => { db = new OrchestrationDb(':memory:') const runtime = createMockRuntime() - const task = db.createTask({ spec: 'send-driven completion' }) + const task = db.createTask({ + runId, + spec: 'send-driven completion' + }) const dispatch = createRootDispatch(db, task.id, 'term_a') const msg = db.insertMessage({ + runId, from: 'term_a', to: 'coord', subject: 'Done', @@ -250,7 +263,10 @@ describe('Coordinator', () => { db = new OrchestrationDb(':memory:') const runtime = createMockRuntime() - const task = db.createTask({ spec: 'duplicate completion' }) + const task = db.createTask({ + runId, + spec: 'duplicate completion' + }) const dispatch = createRootDispatch(db, task.id, 'term_a') const payload = JSON.stringify({ taskId: task.id, @@ -258,6 +274,7 @@ describe('Coordinator', () => { outcome: 'succeeded' }) const first = db.insertMessage({ + runId, from: 'term_a', to: 'coord', subject: 'Done', @@ -265,6 +282,7 @@ describe('Coordinator', () => { payload }) db.insertMessage({ + runId, from: 'term_a', to: 'coord', subject: 'Done again', @@ -289,7 +307,7 @@ describe('Coordinator', () => { db = new OrchestrationDb(':memory:') const runtime = createMockRuntime() - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId, spec: 'work' }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -321,7 +339,10 @@ describe('Coordinator', () => { { handle: 'term_b', worktreeId: 'wt1', connected: true, writable: true } ] - const task = db.createTask({ spec: 'risky work' }) + const task = db.createTask({ + runId, + spec: 'risky work' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -339,6 +360,7 @@ describe('Coordinator', () => { const dispatch = db.getDispatchContext(task.id) expect(dispatch).toBeDefined() db.insertMessage({ + runId, from: dispatch?.assignee_handle ?? 'missing-worker', to: 'coord', subject: `Failed attempt ${i + 1}`, @@ -360,7 +382,10 @@ describe('Coordinator', () => { throw new Error('terminal_not_writable') } - const task = db.createTask({ spec: 'cannot dispatch' }) + const task = db.createTask({ + runId, + spec: 'cannot dispatch' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', coordinatorHandle: 'coord', @@ -379,7 +404,10 @@ describe('Coordinator', () => { const runtime = createMockRuntime() runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] - const task = db.createTask({ spec: 'needs approval' }) + const task = db.createTask({ + runId, + spec: 'needs approval' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -398,6 +426,7 @@ describe('Coordinator', () => { const dispatch = db.getDispatchContext(task.id) expect(dispatch).toBeDefined() db.insertMessage({ + runId, from: 'term_a', to: 'coord', subject: 'Need approval', @@ -441,8 +470,12 @@ describe('Coordinator', () => { const runtime = createMockRuntime() runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] - const t1 = db.createTask({ spec: 'first' }) - const t2 = db.createTask({ spec: 'second', deps: [t1.id] }) + const t1 = db.createTask({ runId, spec: 'first' }) + const t2 = db.createTask({ + runId, + spec: 'second', + deps: [t1.id] + }) expect(t2.status).toBe('pending') @@ -492,9 +525,9 @@ describe('Coordinator', () => { { handle: 'term_c', worktreeId: 'wt1', connected: true, writable: true } ] - const t1 = db.createTask({ spec: 'one' }) - const t2 = db.createTask({ spec: 'two' }) - const t3 = db.createTask({ spec: 'three' }) + const t1 = db.createTask({ runId, spec: 'one' }) + const t2 = db.createTask({ runId, spec: 'two' }) + const t3 = db.createTask({ runId, spec: 'three' }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -530,7 +563,7 @@ describe('Coordinator', () => { const runtime = createMockRuntime() // No terminals available so dispatchReadyTasks creates one and we can // drive the stale-scan deterministically via SQL backdating. - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(db, task.id, 'term_stale') // Backdate dispatched_at and last_heartbeat_at beyond the 10-min threshold @@ -569,7 +602,7 @@ describe('Coordinator', () => { const runtime = createMockRuntime() runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(db, task.id, 'term_a') const coordinator = new Coordinator(db, runtime, { @@ -581,6 +614,7 @@ describe('Coordinator', () => { const runPromise = coordinator.run() db.insertMessage({ + runId, from: 'term_a', to: 'coord', subject: 'alive', @@ -606,12 +640,16 @@ describe('Coordinator', () => { const runtime = createMockRuntime() const logs: string[] = [] - const task = db.createTask({ spec: 'retry-sensitive work' }) + const task = db.createTask({ + runId, + spec: 'retry-sensitive work' + }) const staleCtx = createRootDispatch(db, task.id, 'term_old') db.failDispatch(staleCtx.id, 'retry elsewhere') const activeCtx = createRootDispatch(db, task.id, 'term_current') db.insertMessage({ + runId, from: 'term_old', to: 'coord', subject: 'Late done', @@ -663,11 +701,15 @@ describe('Coordinator', () => { const runtime = createMockRuntime() const logs: string[] = [] - const task = db.createTask({ spec: 'owned work' }) + const task = db.createTask({ + runId, + spec: 'owned work' + }) const leafId = '11111111-1111-4111-8111-111111111111' const ctx = createRootDispatch(db, task.id, 'term_owner', `tab_before:${leafId}`) db.insertMessage({ + runId, from: 'term_reminted', to: 'coord', subject: 'Done after restart', @@ -693,7 +735,7 @@ describe('Coordinator', () => { it('can be stopped', async () => { db = new OrchestrationDb(':memory:') const runtime = createMockRuntime() - db.createTask({ spec: 'never finishes' }) + db.createTask({ runId, spec: 'never finishes' }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -723,7 +765,10 @@ describe('Coordinator', () => { recentSubjects: ['fix A', 'fix B', 'fix C'] }) - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ + runId, + spec: 'do the work' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -759,7 +804,10 @@ describe('Coordinator', () => { recentSubjects: ['fix A'] }) - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ + runId, + spec: 'do the work' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -799,7 +847,7 @@ describe('Coordinator', () => { const spec = `Investigate issue #42 allow-stale-base: true` - const task = db.createTask({ spec }) + const task = db.createTask({ runId, spec }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -832,7 +880,10 @@ allow-stale-base: true` runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] runtime.setProbeDrift(null) - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ + runId, + spec: 'do the work' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -861,7 +912,10 @@ allow-stale-base: true` runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] const logs: string[] = [] - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ + runId, + spec: 'do the work' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -892,7 +946,10 @@ allow-stale-base: true` runtime.terminals = [{ handle: 'term_a', worktreeId: 'wt1', connected: true, writable: true }] runtime.throwProbeDrift = new Error('boom') - const task = db.createTask({ spec: 'do the work' }) + const task = db.createTask({ + runId, + spec: 'do the work' + }) const coordinator = new Coordinator(db, runtime, { spec: 'go', @@ -915,53 +972,3 @@ allow-stale-base: true` }) }) }) - -describe('parseAllowStaleBaseFromSpec', () => { - it('matches canonical form on its own line and strips it', () => { - const spec = `Do the work -allow-stale-base: true` - const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) - expect(allowStale).toBe(true) - expect(strippedSpec).toBe('Do the work\n') - expect(strippedSpec).not.toContain('allow-stale-base') - }) - - it('matches case-insensitively', () => { - const spec = `Do the work -Allow-Stale-Base: TRUE` - const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) - expect(allowStale).toBe(true) - expect(strippedSpec).not.toMatch(/[Aa]llow-[Ss]tale-[Bb]ase/) - }) - - it('does not match allow-stale-base: false', () => { - const spec = `Do the work -allow-stale-base: false` - const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) - expect(allowStale).toBe(false) - expect(strippedSpec).toBe(spec) - }) - - it('does not match allow-stale-base: truthy', () => { - const spec = `Do the work -allow-stale-base: truthy` - const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) - expect(allowStale).toBe(false) - expect(strippedSpec).toBe(spec) - }) - - it('does not match the flag embedded inside a sentence', () => { - const spec = 'we allow-stale-base: true sometimes' - const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) - expect(allowStale).toBe(false) - expect(strippedSpec).toBe(spec) - }) - - it('handles the flag as the last line with no trailing newline', () => { - const spec = 'line 1\nallow-stale-base: true' - const { allowStale, strippedSpec } = parseAllowStaleBaseFromSpec(spec) - expect(allowStale).toBe(true) - expect(strippedSpec).toBe('line 1\n') - expect(strippedSpec.endsWith('allow-stale-base: true')).toBe(false) - }) -}) diff --git a/src/main/runtime/orchestration/db-empty-dispatch-shortcircuit.benchmark.test.ts b/src/main/runtime/orchestration/db-empty-dispatch-shortcircuit.benchmark.test.ts index c729a8b1dd7..6407e67bc4b 100644 --- a/src/main/runtime/orchestration/db-empty-dispatch-shortcircuit.benchmark.test.ts +++ b/src/main/runtime/orchestration/db-empty-dispatch-shortcircuit.benchmark.test.ts @@ -64,7 +64,7 @@ describe('orchestration empty-dispatch short-circuit (benchmark)', () => { it('still runs the fan-out once a dispatch exists (correctness preserved)', () => { const db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) createRootDispatch(db, task.id, 'term_5') const handles = Array.from({ length: 10 }, (_, i) => `term_${i}`) @@ -76,7 +76,11 @@ describe('orchestration empty-dispatch short-circuit (benchmark)', () => { it('predicate lifecycle: false when empty, true after dispatch (even completed), false after reset', () => { const db = new OrchestrationDb(':memory:') expect(db.hasAnyDispatchContexts()).toBe(false) - const ctx = createRootDispatch(db, db.createTask({ spec: 'work' }).id, 'term_worker') + const ctx = createRootDispatch( + db, + db.createTask({ runId: 'run_legacy_local', spec: 'work' }).id, + 'term_worker' + ) expect(db.hasAnyDispatchContexts()).toBe(true) // Completed rows still count — recent-completed lookups must stay valid. db.completeDispatch(ctx.id) diff --git a/src/main/runtime/orchestration/db-heartbeat-straggler-guard.test.ts b/src/main/runtime/orchestration/db-heartbeat-straggler-guard.test.ts index 793c218e162..c7743757aa9 100644 --- a/src/main/runtime/orchestration/db-heartbeat-straggler-guard.test.ts +++ b/src/main/runtime/orchestration/db-heartbeat-straggler-guard.test.ts @@ -13,7 +13,7 @@ afterEach(() => { function seedHeartbeatedDispatch(): { d: OrchestrationDb; dispatchId: string } { const d = new OrchestrationDb(':memory:') db = d - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(d, task.id, 'term_worker') d.recordHeartbeat(dispatch.id, '2026-05-03T00:00:00.000Z') return { d, dispatchId: dispatch.id } diff --git a/src/main/runtime/orchestration/db-message-timestamp.test.ts b/src/main/runtime/orchestration/db-message-timestamp.test.ts index d4d000c48d3..3900904e5b1 100644 --- a/src/main/runtime/orchestration/db-message-timestamp.test.ts +++ b/src/main/runtime/orchestration/db-message-timestamp.test.ts @@ -8,7 +8,12 @@ describe('orchestration message timestamps', () => { it('exposes SQLite timestamps with an explicit UTC designator', () => { db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'b', subject: 'timestamped' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'b', + subject: 'timestamped' + }) expect(message.created_at).toMatch(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?Z$/) db.markAsDelivered([message.id]) diff --git a/src/main/runtime/orchestration/db-messages.test.ts b/src/main/runtime/orchestration/db-messages.test.ts new file mode 100644 index 00000000000..789797d12f0 --- /dev/null +++ b/src/main/runtime/orchestration/db-messages.test.ts @@ -0,0 +1,194 @@ +import { afterEach, describe, expect, it } from 'vitest' +import type Database from '../../sqlite/sync-database' +import { OrchestrationDb, type MessageType } from './db' + +const runId = 'run_legacy_local' + +describe('OrchestrationDb', () => { + let db: OrchestrationDb | undefined + + afterEach(() => { + db?.close() + }) + + function createDb(): OrchestrationDb { + db = new OrchestrationDb(':memory:') + return db + } + + describe('messages', () => { + it('inserts and retrieves a message', () => { + const d = createDb() + const msg = d.insertMessage({ + runId, + from: 'term_a', + to: 'term_b', + subject: 'hello', + body: 'world' + }) + expect(msg.id).toMatch(/^msg_/) + expect(msg.from_handle).toBe('term_a') + expect(msg.to_handle).toBe('term_b') + expect(msg.subject).toBe('hello') + expect(msg.body).toBe('world') + expect(msg.type).toBe('status') + expect(msg.priority).toBe('normal') + expect(msg.read).toBe(0) + expect(msg.sequence).toBeGreaterThan(0) + }) + + it('returns unread messages in sequence order', () => { + const d = createDb() + d.insertMessage({ runId, from: 'a', to: 'b', subject: 'first' }) + d.insertMessage({ runId, from: 'a', to: 'b', subject: 'second' }) + d.insertMessage({ runId, from: 'a', to: 'c', subject: 'other' }) + + const unread = d.getUnreadMessages('b') + expect(unread).toHaveLength(2) + expect(unread[0].subject).toBe('first') + expect(unread[1].subject).toBe('second') + }) + + it('filters unread by type', () => { + const d = createDb() + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'status msg', + type: 'status' + }) + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'done msg', + type: 'worker_done' + }) + + const filtered = d.getUnreadMessages('b', ['worker_done']) + expect(filtered).toHaveLength(1) + expect(filtered[0].type).toBe('worker_done') + }) + + it('excludes already-delivered rows from getUndeliveredUnreadMessages', () => { + const d = createDb() + const m1 = d.insertMessage({ runId, from: 'a', to: 'b', subject: 'one' }) + const m2 = d.insertMessage({ runId, from: 'a', to: 'b', subject: 'two' }) + + d.markAsDelivered([m1.id]) + + // Push delivery query: only undelivered, unread. + const pending = d.getUndeliveredUnreadMessages('b') + expect(pending).toHaveLength(1) + expect(pending[0].id).toBe(m2.id) + + // Explicit `check` still sees both (they are still unread). + const unread = d.getUnreadMessages('b') + expect(unread).toHaveLength(2) + }) + + it('creates the undelivered inbox index used by push delivery', () => { + const d = createDb() + const sqlite = (d as unknown as { db: Database.Database }).db + + const indexes = sqlite + .prepare( + `SELECT name FROM sqlite_master WHERE type = 'index' AND tbl_name = 'messages' AND name = 'idx_messages_undelivered_inbox'` + ) + .all() + + expect(indexes).toHaveLength(1) + }) + + it('filters getUndeliveredUnreadMessages by type', () => { + const d = createDb() + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 's', + type: 'status' + }) + const wd = d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'd', + type: 'worker_done' + }) + + const filtered = d.getUndeliveredUnreadMessages('b', ['worker_done']) + expect(filtered).toHaveLength(1) + expect(filtered[0].id).toBe(wd.id) + }) + + it('marks messages as read', () => { + const d = createDb() + const m1 = d.insertMessage({ runId, from: 'a', to: 'b', subject: 'one' }) + const m2 = d.insertMessage({ runId, from: 'a', to: 'b', subject: 'two' }) + + d.markAsRead([m1.id]) + + const unread = d.getUnreadMessages('b') + expect(unread).toHaveLength(1) + expect(unread[0].id).toBe(m2.id) + }) + + it('stores typed payload and thread_id', () => { + const d = createDb() + const payload = JSON.stringify({ taskId: 'task_abc', filesModified: ['src/a.ts'] }) + const msg = d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'done', + type: 'worker_done', + priority: 'high', + threadId: 'thread_1', + payload + }) + + expect(msg.type).toBe('worker_done') + expect(msg.priority).toBe('high') + expect(msg.thread_id).toBe('thread_1') + expect(msg.payload).toBe(payload) + }) + + it('rejects invalid message type', () => { + const d = createDb() + expect(() => + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'bad', + type: 'invalid' as MessageType + }) + ).toThrow() + }) + + it('getInbox returns all messages across recipients', () => { + const d = createDb() + d.insertMessage({ runId, from: 'a', to: 'b', subject: 'one' }) + d.insertMessage({ runId, from: 'a', to: 'c', subject: 'two' }) + d.insertMessage({ runId, from: 'b', to: 'a', subject: 'three' }) + + const inbox = d.getInbox(10) + expect(inbox).toHaveLength(3) + }) + + it('getMessageById returns the correct message', () => { + const d = createDb() + const msg = d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'test' + }) + const found = d.getMessageById(msg.id) + expect(found?.subject).toBe('test') + expect(d.getMessageById('msg_nonexistent')).toBeUndefined() + }) + }) +}) diff --git a/src/main/runtime/orchestration/db-stopping-worker-task-guard.test.ts b/src/main/runtime/orchestration/db-stopping-worker-task-guard.test.ts new file mode 100644 index 00000000000..a747ae07a51 --- /dev/null +++ b/src/main/runtime/orchestration/db-stopping-worker-task-guard.test.ts @@ -0,0 +1,122 @@ +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { OrchestrationDb } from './db' +import { createRootDispatch } from './db/root-dispatch-test-fixture' + +const PANE_W = 'tab_w:aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa' + +describe('a Task whose supervised worker is stopping', () => { + let db: OrchestrationDb + beforeEach(() => { + db = new OrchestrationDb(':memory:') + }) + afterEach(() => db.close()) + + function localWorker() { + const task = db.createTask({ runId: 'run_legacy_local', spec: 'local work' }) + const { dispatch } = db.createStartingWorkerDispatch({ + taskId: task.id, + startOptions: {}, + creator: { kind: 'system' }, + maxDepth: 9 + }) + db.prepareStartingWorkerAuthority({ + dispatchId: dispatch.id, + handle: 'term_w', + paneKey: PANE_W, + processIncarnation: 'inc1', + worktreeId: 'wt', + effects: [], + setupState: 'not_configured' + }) + db.markWorkerDispatchReady(dispatch.id) + return { task, dispatch } + } + + describe('task-update', () => { + it('refuses to re-open the Task while the worker is stopping', () => { + const { task, dispatch } = localWorker() + db.beginWorkerStop(dispatch.id, 'epoch_home') + expect(db.getTask(task.id)?.status).toBe('blocked') + + expect(() => db.updateTaskStatus(task.id, 'dispatched')).toThrowError( + expect.objectContaining({ + code: 'task_not_startable', + data: { taskId: task.id, dispatchId: dispatch.id } + }) + ) + expect(db.getTask(task.id)?.status).toBe('blocked') + }) + + it('refuses to re-open the Task while the stop outcome is unknown', () => { + const { task, dispatch } = localWorker() + db.beginWorkerStop(dispatch.id, 'epoch_home') + db.markWorkerStopUnknown(dispatch.id, 'the execution host did not answer') + + expect(() => db.updateTaskStatus(task.id, 'dispatched')).toThrowError( + expect.objectContaining({ code: 'task_not_startable' }) + ) + expect(db.getTask(task.id)?.status).toBe('blocked') + }) + + it('control: still accepts dispatched for an active Dispatch with no supervised worker', () => { + const task = db.createTask({ runId: 'run_legacy_local', spec: 'unsupervised work' }) + createRootDispatch(db, task.id, 'term_worker') + + expect(db.updateTaskStatus(task.id, 'dispatched')?.status).toBe('dispatched') + }) + + it('control: still accepts dispatched while the supervised worker is ready', () => { + const { task } = localWorker() + + expect(db.updateTaskStatus(task.id, 'dispatched')?.status).toBe('dispatched') + }) + + it('control: a no-op re-assert of dispatched under a stopping worker stays legal', () => { + const { task, dispatch } = localWorker() + expect(db.getTask(task.id)?.status).toBe('dispatched') + db.beginWorkerStop(dispatch.id, 'epoch_home') + // beginWorkerStop moved the Task to blocked; put it back the only way that is not a re-open. + db.db.prepare("UPDATE tasks SET status = 'dispatched' WHERE id = ?").run(task.id) + + expect(db.updateTaskStatus(task.id, 'dispatched')?.status).toBe('dispatched') + }) + }) + + describe('operator escape', () => { + it('accepts a re-issued worker-stop and reaches an honest stop_unknown outcome', () => { + const { task, dispatch } = localWorker() + db.beginWorkerStop(dispatch.id, 'epoch_dead_runtime') + + // The runtime that owned the first stop died mid-flight; the re-issue is the way out. + const reissued = db.beginWorkerStop(dispatch.id, 'epoch_new_runtime') + expect(reissued).toMatchObject({ disposition: 'stopping' }) + expect(db.getWorkerDispatch(dispatch.id)?.runtime_epoch).toBe('epoch_new_runtime') + + db.markWorkerStopUnknown(dispatch.id, 'the execution host did not answer') + expect(db.abandonWorkerDispatch(dispatch.id)).toMatchObject({ disposition: 'abandoned' }) + expect(db.getTask(task.id)?.status).toBe('blocked') + }) + + it('refuses a re-issue from the runtime whose own stop is still in flight', () => { + const { dispatch } = localWorker() + db.beginWorkerStop(dispatch.id, 'epoch_this_runtime') + + // The terminal is closing and its exit event has not landed yet. Letting this second pass + // record stop_unknown would make the exit read as a crash instead of this stop succeeding. + expect(() => db.beginWorkerStop(dispatch.id, 'epoch_this_runtime')).toThrowError( + /cannot stop from stopping/ + ) + + // The row is still the one the exit path claims a clean stop from: stopping, same epoch. + expect(db.getWorkerDispatch(dispatch.id)).toMatchObject({ + state: 'stopping', + runtime_epoch: 'epoch_this_runtime' + }) + expect(db.settleWorkerStop(dispatch.id).state).toBe('stopped') + expect(db.getDispatchContextById(dispatch.id)).toMatchObject({ + status: 'failed', + last_failure: 'stopped' + }) + }) + }) +}) diff --git a/src/main/runtime/orchestration/db-task-create-readiness.test.ts b/src/main/runtime/orchestration/db-task-create-readiness.test.ts index 019b627da6c..4b661829d02 100644 --- a/src/main/runtime/orchestration/db-task-create-readiness.test.ts +++ b/src/main/runtime/orchestration/db-task-create-readiness.test.ts @@ -33,12 +33,16 @@ describe('task creation dependency readiness', () => { it('creates a late dependent as ready when every dependency is completed', () => { const db = createDb() - const first = db.createTask({ spec: 'first' }) - const second = db.createTask({ spec: 'second' }) + const first = db.createTask({ runId: 'run_legacy_local', spec: 'first' }) + const second = db.createTask({ runId: 'run_legacy_local', spec: 'second' }) db.updateTaskStatus(first.id, 'completed') db.updateTaskStatus(second.id, 'completed') - const child = db.createTask({ spec: 'child', deps: [first.id, second.id] }) + const child = db.createTask({ + runId: 'run_legacy_local', + spec: 'child', + deps: [first.id, second.id] + }) expect(child.status).toBe('ready') }) @@ -49,7 +53,7 @@ describe('task creation dependency readiness', () => { const path = join(directory, 'orchestration.db') const db = createDb(path) const concurrent = createDb(path) - const dependency = db.createTask({ spec: 'dependency' }) + const dependency = db.createTask({ runId: 'run_legacy_local', spec: 'dependency' }) const sqlite = (db as unknown as OrchestrationDbAccess).db const prepare = sqlite.prepare.bind(sqlite) let injected = false @@ -61,7 +65,7 @@ describe('task creation dependency readiness', () => { return prepare(sql) }) - const child = db.createTask({ spec: 'child', deps: [dependency.id] }) + const child = db.createTask({ runId: 'run_legacy_local', spec: 'child', deps: [dependency.id] }) expect(injected).toBe(true) expect(child.status).toBe('ready') @@ -69,10 +73,14 @@ describe('task creation dependency readiness', () => { it('promotes only after every dependency completes', () => { const db = createDb() - const first = db.createTask({ spec: 'first' }) - const second = db.createTask({ spec: 'second' }) + const first = db.createTask({ runId: 'run_legacy_local', spec: 'first' }) + const second = db.createTask({ runId: 'run_legacy_local', spec: 'second' }) db.updateTaskStatus(first.id, 'completed') - const child = db.createTask({ spec: 'child', deps: [first.id, second.id] }) + const child = db.createTask({ + runId: 'run_legacy_local', + spec: 'child', + deps: [first.id, second.id] + }) expect(child.status).toBe('pending') db.updateTaskStatus(second.id, 'completed') @@ -83,11 +91,15 @@ describe('task creation dependency readiness', () => { 'does not unlock a dependent whose dependency is %s', (status) => { const db = createDb() - const terminal = db.createTask({ spec: 'terminal dependency' }) - const completing = db.createTask({ spec: 'completing dependency' }) + const terminal = db.createTask({ runId: 'run_legacy_local', spec: 'terminal dependency' }) + const completing = db.createTask({ runId: 'run_legacy_local', spec: 'completing dependency' }) db.updateTaskStatus(terminal.id, status) - const child = db.createTask({ spec: 'child', deps: [terminal.id, completing.id] }) + const child = db.createTask({ + runId: 'run_legacy_local', + spec: 'child', + deps: [terminal.id, completing.id] + }) expect(child.status).toBe('pending') db.updateTaskStatus(completing.id, 'completed') @@ -98,9 +110,9 @@ describe('task creation dependency readiness', () => { it('rejects missing dependencies without inserting a task', () => { const db = createDb() - expect(() => db.createTask({ spec: 'child', deps: ['task_missing'] })).toThrow( - 'Dependency task task_missing must belong to run' - ) + expect(() => + db.createTask({ runId: 'run_legacy_local', spec: 'child', deps: ['task_missing'] }) + ).toThrow('Dependency task task_missing must belong to run') expect(db.listTasks()).toEqual([]) }) @@ -109,7 +121,7 @@ describe('task creation dependency readiness', () => { const sqlite = (db as unknown as OrchestrationDbAccess).db sqlite.exec('BEGIN IMMEDIATE') - const task = db.createTask({ spec: 'transactional child' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'transactional child' }) sqlite.exec('ROLLBACK') expect(db.getTask(task.id)).toBeUndefined() @@ -120,11 +132,19 @@ describe('task creation dependency readiness', () => { directories.push(directory) const path = join(directory, 'orchestration.db') const before = createDb(path) - const completed = before.createTask({ spec: 'completed' }) - const open = before.createTask({ spec: 'open' }) + const completed = before.createTask({ runId: 'run_legacy_local', spec: 'completed' }) + const open = before.createTask({ runId: 'run_legacy_local', spec: 'open' }) before.updateTaskStatus(completed.id, 'completed') - const ready = before.createTask({ spec: 'ready', deps: [completed.id] }) - const pending = before.createTask({ spec: 'pending', deps: [completed.id, open.id] }) + const ready = before.createTask({ + runId: 'run_legacy_local', + spec: 'ready', + deps: [completed.id] + }) + const pending = before.createTask({ + runId: 'run_legacy_local', + spec: 'pending', + deps: [completed.id, open.id] + }) before.close() databases.splice(databases.indexOf(before), 1) diff --git a/src/main/runtime/orchestration/db-task-dispatch-invariant.test.ts b/src/main/runtime/orchestration/db-task-dispatch-invariant.test.ts index 53340b6dce5..2b81b2f2897 100644 --- a/src/main/runtime/orchestration/db-task-dispatch-invariant.test.ts +++ b/src/main/runtime/orchestration/db-task-dispatch-invariant.test.ts @@ -30,9 +30,17 @@ describe('Task/Dispatch invariant transactions', () => { 'allows a dependency-blocked pending Task to become %s', (status) => { const { db } = createDatabase() - const dependency = db.createTask({ spec: 'unresolved dependency' }) - const task = db.createTask({ spec: 'manual resolution', deps: [dependency.id] }) - const dependent = db.createTask({ spec: 'downstream work', deps: [task.id] }) + const dependency = db.createTask({ runId: 'run_legacy_local', spec: 'unresolved dependency' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'manual resolution', + deps: [dependency.id] + }) + const dependent = db.createTask({ + runId: 'run_legacy_local', + spec: 'downstream work', + deps: [task.id] + }) expect(task.status).toBe('pending') const updated = db.updateTaskStatus(task.id, status, 'manual resolution') @@ -45,7 +53,7 @@ describe('Task/Dispatch invariant transactions', () => { it('surfaces invalid Task lifecycle edges instead of returning the unchanged row', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'invalid lifecycle edge' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'invalid lifecycle edge' }) db.updateTaskStatus(task.id, 'blocked') expect(() => @@ -67,8 +75,12 @@ describe('Task/Dispatch invariant transactions', () => { 'rolls back a %s Task when Dispatch settlement fails', (status) => { const { db } = createDatabase() - const task = db.createTask({ spec: 'atomic work' }) - const dependent = db.createTask({ spec: 'dependent work', deps: [task.id] }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'atomic work' }) + const dependent = db.createTask({ + runId: 'run_legacy_local', + spec: 'dependent work', + deps: [task.id] + }) const dispatch = createRootDispatch(db, task.id, 'term_worker') const capability = db.mintDispatchCapability({ dispatchId: dispatch.id, @@ -111,7 +123,10 @@ describe('Task/Dispatch invariant transactions', () => { it('does not commit a caller-owned transaction', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'outer transaction work' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'outer transaction work' + }) const dispatch = createRootDispatch(db, task.id, 'term_worker') const sqlite = sqliteFor(db) @@ -135,7 +150,10 @@ describe('Task/Dispatch invariant transactions', () => { it('keeps Dispatch creation inside a caller-owned transaction', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'outer transaction dispatch' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'outer transaction dispatch' + }) const sqlite = sqliteFor(db) sqlite.exec('BEGIN IMMEDIATE') @@ -152,7 +170,10 @@ describe('Task/Dispatch invariant transactions', () => { 'settles every active Dispatch left by a pre-fix split when the Task becomes %s', (status) => { const { db } = createDatabase() - const task = db.createTask({ spec: 'legacy split work' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'legacy split work' + }) const first = createRootDispatch(db, task.id, 'term_first') sqliteFor(db).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const second = createRootDispatch(db, task.id, 'term_second') @@ -169,17 +190,31 @@ describe('Task/Dispatch invariant transactions', () => { expect(db.getActiveDispatchForTerminal('term_first')).toBeUndefined() expect(db.getActiveDispatchForTerminal('term_second')).toBeUndefined() expect(() => - createRootDispatch(db, db.createTask({ spec: 'first later work' }).id, 'term_first') + createRootDispatch( + db, + db.createTask({ runId: 'run_legacy_local', spec: 'first later work' }).id, + 'term_first' + ) ).not.toThrow() expect(() => - createRootDispatch(db, db.createTask({ spec: 'second later work' }).id, 'term_second') + createRootDispatch( + db, + db.createTask({ + runId: 'run_legacy_local', + spec: 'second later work' + }).id, + 'term_second' + ) ).not.toThrow() } ) it('does not requeue a legacy split Task while another Dispatch remains active', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'legacy split retry' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'legacy split retry' + }) const first = createRootDispatch(db, task.id, 'term_first') sqliteFor(db).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const second = createRootDispatch(db, task.id, 'term_second') @@ -193,7 +228,10 @@ describe('Task/Dispatch invariant transactions', () => { it('does not block a legacy split Task while another Dispatch remains active', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'legacy split release' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'legacy split release' + }) const first = createRootDispatch(db, task.id, 'term_first') sqliteFor(db).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const second = createRootDispatch(db, task.id, 'term_second') @@ -211,7 +249,10 @@ describe('Task/Dispatch invariant transactions', () => { 'rejects moving a Task to %s while a Dispatch remains active', (status) => { const { db } = createDatabase() - const task = db.createTask({ spec: 'guarded work' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'guarded work' + }) const dispatch = createRootDispatch(db, task.id, 'term_worker') expect(() => db.updateTaskStatus(task.id, status, 'must not persist')).toThrowError( @@ -227,7 +268,10 @@ describe('Task/Dispatch invariant transactions', () => { it('rejects moving a Task to dispatched without an active Dispatch', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'unassigned work' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'unassigned work' + }) expect(() => db.updateTaskStatus(task.id, 'dispatched')).toThrowError( expect.objectContaining({ @@ -241,7 +285,10 @@ describe('Task/Dispatch invariant transactions', () => { it('rejects a Dispatch when failure wins after readiness was observed', () => { const first = createDatabase() const concurrent = createDatabase(first.path) - const task = first.db.createTask({ spec: 'interleaved work' }) + const task = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'interleaved work' + }) const sqlite = sqliteFor(first.db) const prepare = sqlite.prepare.bind(sqlite) let injected = false @@ -264,8 +311,14 @@ describe('Task/Dispatch invariant transactions', () => { it('atomically rejects a same-pane Dispatch that loses the occupancy race', () => { const first = createDatabase() const concurrent = createDatabase(first.path) - const firstTask = first.db.createTask({ spec: 'first terminal claimant' }) - const secondTask = first.db.createTask({ spec: 'second terminal claimant' }) + const firstTask = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'first terminal claimant' + }) + const secondTask = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'second terminal claimant' + }) const sqlite = sqliteFor(first.db) const prepare = sqlite.prepare.bind(sqlite) let winnerId: string | undefined @@ -303,14 +356,20 @@ describe('Task/Dispatch invariant transactions', () => { it('rejects worker authority when another Dispatch owns the pane', () => { const { db } = createDatabase() - const ownerTask = db.createTask({ spec: 'current pane owner' }) + const ownerTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'current pane owner' + }) const owner = createRootDispatch( db, ownerTask.id, 'term_owner', 'tab_old:cccccccc-cccc-4ccc-8ccc-cccccccccccc' ) - const workerTask = db.createTask({ spec: 'competing supervised worker' }) + const workerTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'competing supervised worker' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -346,7 +405,10 @@ describe('Task/Dispatch invariant transactions', () => { 'rejects a %s Task update while its supervised worker remains active', (status) => { const { db } = createDatabase() - const task = db.createTask({ spec: 'supervised lifecycle' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'supervised lifecycle' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -394,7 +456,10 @@ describe('Task/Dispatch invariant transactions', () => { it('keeps a federated late start authoritative after rejecting Task failure', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'federated lifecycle' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'federated lifecycle' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/db-task-dispatch-lifecycle-guards.test.ts b/src/main/runtime/orchestration/db-task-dispatch-lifecycle-guards.test.ts index bb6d3472d4e..f4cdcdf63b8 100644 --- a/src/main/runtime/orchestration/db-task-dispatch-lifecycle-guards.test.ts +++ b/src/main/runtime/orchestration/db-task-dispatch-lifecycle-guards.test.ts @@ -29,7 +29,7 @@ afterEach(() => { describe('Task/Dispatch lifecycle guards', () => { it('rejects a worker report while another supervised Dispatch is active', () => { const database = createDatabase() - const task = database.createTask({ spec: 'legacy supervised split' }) + const task = database.createTask({ runId: 'run_legacy_local', spec: 'legacy supervised split' }) const first = startWorker(database, task.id, 'first') sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const second = startWorker(database, task.id, 'second') @@ -55,7 +55,7 @@ describe('Task/Dispatch lifecycle guards', () => { 'settles context-only legacy siblings after a %s worker report', (outcome) => { const database = createDatabase() - const task = database.createTask({ spec: 'legacy mixed split' }) + const task = database.createTask({ runId: 'run_legacy_local', spec: 'legacy mixed split' }) const contextOnly = createRootDispatch(database, task.id, 'term_context') sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const worker = startWorker(database, task.id, 'reporter') @@ -78,7 +78,10 @@ describe('Task/Dispatch lifecycle guards', () => { expect(() => createRootDispatch( database, - database.createTask({ spec: 'later context work' }).id, + database.createTask({ + runId: 'run_legacy_local', + spec: 'later context work' + }).id, 'term_context' ) ).not.toThrow() @@ -87,7 +90,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('settles a newer context-only legacy sibling after a worker report', () => { const database = createDatabase() - const task = database.createTask({ spec: 'reversed legacy mixed split' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'reversed legacy mixed split' + }) const worker = startWorker(database, task.id, 'reversed_reporter') sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const contextOnly = createRootDispatch(database, task.id, 'term_reversed_context') @@ -112,7 +118,10 @@ describe('Task/Dispatch lifecycle guards', () => { 'treats abandon of an already %s worker as stale without a lifecycle conflict', (state) => { const database = createDatabase() - const task = database.createTask({ spec: `already ${state}` }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: `already ${state}` + }) const worker = startWorker(database, task.id, `already_${state}`) if (state === 'failed') { database.failDispatch(worker.dispatchId, 'process exited', { workerProcessExited: true }) @@ -130,7 +139,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('rejects generic failure while a supervised worker remains active', () => { const database = createDatabase() - const task = database.createTask({ spec: 'supervised failure guard' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'supervised failure guard' + }) const worker = startWorker(database, task.id, 'guarded') expect(() => database.failDispatch(worker.dispatchId, 'unsafe retry')).toThrowError( @@ -151,7 +163,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('atomically settles worker state when a proven process exit fails its Dispatch', () => { const database = createDatabase() - const task = database.createTask({ spec: 'exited worker' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'exited worker' + }) const worker = startWorker(database, task.id, 'exited') expect( @@ -168,7 +183,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('settles a stop-unknown worker when a positive PTY exit arrives', () => { const database = createDatabase() - const task = database.createTask({ spec: 'stop-unknown exited worker' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'stop-unknown exited worker' + }) const worker = startWorker(database, task.id, 'stop_unknown_exited') expect(database.beginWorkerStop(worker.dispatchId, 'runtime_test').disposition).toBe('stopping') @@ -198,7 +216,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('keeps a Task dispatched when missing-terminal recovery leaves another worker active', () => { const database = createDatabase() - const task = database.createTask({ spec: 'legacy missing-terminal split' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'legacy missing-terminal split' + }) const missing = startWorker(database, task.id, 'missing') sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const live = startWorker(database, task.id, 'live') @@ -225,7 +246,10 @@ describe('Task/Dispatch lifecycle guards', () => { 'keeps a Task dispatched when a %s worker start fails beside a live worker', (kind) => { const database = createDatabase() - const task = database.createTask({ spec: `${kind} split start failure` }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: `${kind} split start failure` + }) const failed = database.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -342,7 +366,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('rolls back federated start uncertainty when the Task transition cannot commit', () => { const database = createDatabase() - const task = database.createTask({ spec: 'atomic federated uncertainty' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'atomic federated uncertainty' + }) const started = database.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -383,7 +410,10 @@ describe('Task/Dispatch lifecycle guards', () => { '%s releases the last context-only sibling after a newer worker start fails', (operation) => { const database = createDatabase() - const task = database.createTask({ spec: `${operation} historical sibling` }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: `${operation} historical sibling` + }) const contextOnly = createRootDispatch(database, task.id, `term_${operation}`) sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const failed = database.createStartingWorkerDispatch({ @@ -411,7 +441,10 @@ describe('Task/Dispatch lifecycle guards', () => { expect(() => createRootDispatch( database, - database.createTask({ spec: `${operation} later work` }).id, + database.createTask({ + runId: 'run_legacy_local', + spec: `${operation} later work` + }).id, `term_${operation}` ) ).not.toThrow() @@ -422,7 +455,10 @@ describe('Task/Dispatch lifecycle guards', () => { '%s records guarded receipts for context-only Dispatch and Task release', (operation) => { const database = createDatabase() - const task = database.createTask({ spec: `${operation} receipt release` }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: `${operation} receipt release` + }) const contextOnly = createRootDispatch(database, task.id, `term_${operation}`) const released = @@ -445,7 +481,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('rolls back both context-only projections when the Task transition fails', () => { const database = createDatabase() - const task = database.createTask({ spec: 'context-only atomic receipt' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'context-only atomic receipt' + }) const contextOnly = createRootDispatch(database, task.id, 'term_context') sqliteFor(database).exec(` CREATE TRIGGER reject_context_release_task_block @@ -470,7 +509,10 @@ describe('Task/Dispatch lifecycle guards', () => { '%s preserves a live worker sibling and lets it report', (operation) => { const database = createDatabase() - const task = database.createTask({ spec: `${operation} legacy worker split` }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: `${operation} legacy worker split` + }) const live = startWorker(database, task.id, `${operation}_live`) sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const released = startWorker(database, task.id, `${operation}_released`) @@ -502,7 +544,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('blocks a Task when an interleaved stop settles its final active Dispatch', () => { const database = createDatabase() - const task = database.createTask({ spec: 'interleaved legacy worker release' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'interleaved legacy worker release' + }) const stopping = startWorker(database, task.id, 'interleaved_stopping') sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const abandoned = startWorker(database, task.id, 'interleaved_abandoned') @@ -521,7 +566,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('restores a live sibling after stopping an uncertain worker start', () => { const database = createDatabase() - const task = database.createTask({ spec: 'uncertain legacy worker split' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'uncertain legacy worker split' + }) const live = startWorker(database, task.id, 'uncertain_live') sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const uncertain = database.createStartingWorkerDispatch({ @@ -552,7 +600,10 @@ describe('Task/Dispatch lifecycle guards', () => { 'restores a live sibling after an uncertain worker start fails through %s', (recovery) => { const database = createDatabase() - const task = database.createTask({ spec: `${recovery} uncertain sibling` }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: `${recovery} uncertain sibling` + }) const live = startWorker(database, task.id, `${recovery}_live`) sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const uncertain = database.createStartingWorkerDispatch({ @@ -589,7 +640,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('rejects gate creation while a supervised worker remains active', () => { const database = createDatabase() - const task = database.createTask({ spec: 'worker gate guard' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'worker gate guard' + }) const worker = startWorker(database, task.id, 'gate') expect(() => database.createGate({ taskId: task.id, question: 'Proceed?' })).toThrowError( @@ -607,7 +661,10 @@ describe('Task/Dispatch lifecycle guards', () => { it('rolls back gate resolution when an active Dispatch blocks readiness', () => { const database = createDatabase() - const task = database.createTask({ spec: 'corrupt gated task' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'corrupt gated task' + }) const gate = database.createGate({ taskId: task.id, question: 'Proceed?' }) sqliteFor(database).prepare("UPDATE tasks SET status = 'ready' WHERE id = ?").run(task.id) const dispatch = createRootDispatch(database, task.id, 'term_worker') diff --git a/src/main/runtime/orchestration/db-task-dispatch-races.test.ts b/src/main/runtime/orchestration/db-task-dispatch-races.test.ts index b4c27ac0c64..c671aa4566b 100644 --- a/src/main/runtime/orchestration/db-task-dispatch-races.test.ts +++ b/src/main/runtime/orchestration/db-task-dispatch-races.test.ts @@ -29,7 +29,10 @@ describe('Task/Dispatch concurrency', () => { it('reads a concurrent Task result before applying an explicit status correction', () => { const first = createDatabase() const concurrent = createDatabase(first.path) - const task = first.db.createTask({ spec: 'concurrent status winner' }) + const task = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'concurrent status winner' + }) const sqlite = sqliteFor(first.db) const exec = sqlite.exec.bind(sqlite) let concurrentWon = false @@ -57,7 +60,7 @@ describe('Task/Dispatch concurrency', () => { it('holds the Task status writer reservation through its lifecycle reads', () => { const first = createDatabase() const concurrent = createDatabase(first.path) - const task = first.db.createTask({ spec: 'reserved status winner' }) + const task = first.db.createTask({ runId: 'run_legacy_local', spec: 'reserved status winner' }) const sqlite = sqliteFor(first.db) const exec = sqlite.exec.bind(sqlite) sqliteFor(concurrent.db).pragma('busy_timeout = 0') @@ -86,7 +89,7 @@ describe('Task/Dispatch concurrency', () => { it('rolls back Dispatch failure when Task requeue fails', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'atomic retry failure' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'atomic retry failure' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') sqliteFor(db).exec(` CREATE TRIGGER reject_task_requeue @@ -113,7 +116,10 @@ describe('Task/Dispatch concurrency', () => { it('does not let stale failure overwrite a completed worker report', () => { const first = createDatabase() const concurrent = createDatabase(first.path) - const task = first.db.createTask({ spec: 'worker completion wins' }) + const task = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'worker completion wins' + }) const started = first.db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -177,7 +183,10 @@ describe('Task/Dispatch concurrency', () => { it('keeps nested dispatch failure atomic with its caller transaction', () => { const { db } = createDatabase() - const task = db.createTask({ spec: 'nested atomic failure' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'nested atomic failure' + }) const dispatch = createRootDispatch(db, task.id, 'term_worker') const sqlite = sqliteFor(db) @@ -198,8 +207,14 @@ describe('Task/Dispatch concurrency', () => { it('serializes reminted-pane worker authority claims', () => { const first = createDatabase() const concurrent = createDatabase(first.path) - const losingTask = first.db.createTask({ spec: 'losing worker' }) - const winningTask = first.db.createTask({ spec: 'winning worker' }) + const losingTask = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'losing worker' + }) + const winningTask = first.db.createTask({ + runId: 'run_legacy_local', + spec: 'winning worker' + }) const loser = first.db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/db-undelivered-mailboxes.test.ts b/src/main/runtime/orchestration/db-undelivered-mailboxes.test.ts index 86bc9fdf46b..b20dd625310 100644 --- a/src/main/runtime/orchestration/db-undelivered-mailboxes.test.ts +++ b/src/main/runtime/orchestration/db-undelivered-mailboxes.test.ts @@ -8,10 +8,20 @@ describe('undelivered orchestration mailboxes', () => { it('lists only mailboxes with undelivered unread messages', () => { db = new OrchestrationDb(':memory:') - const delivered = db.insertMessage({ from: 'a', to: 'delivered', subject: 'done' }) - const read = db.insertMessage({ from: 'a', to: 'read', subject: 'seen' }) - db.insertMessage({ from: 'a', to: 'pending', subject: 'first' }) - db.insertMessage({ from: 'a', to: 'pending', subject: 'second' }) + const delivered = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'delivered', + subject: 'done' + }) + const read = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'read', + subject: 'seen' + }) + db.insertMessage({ runId: 'run_legacy_local', from: 'a', to: 'pending', subject: 'first' }) + db.insertMessage({ runId: 'run_legacy_local', from: 'a', to: 'pending', subject: 'second' }) db.markAsDelivered([delivered.id]) db.markAsRead([read.id]) @@ -20,7 +30,12 @@ describe('undelivered orchestration mailboxes', () => { it('persists and settles a pending pointer Enter independently of delivery', () => { db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'run:run_1', subject: 'staged' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run_1', + subject: 'staged' + }) expect( db.stageMailboxPointerEnter([message.id], { diff --git a/src/main/runtime/orchestration/db.test.ts b/src/main/runtime/orchestration/db.test.ts index 4825246586a..33af326f34d 100644 --- a/src/main/runtime/orchestration/db.test.ts +++ b/src/main/runtime/orchestration/db.test.ts @@ -4,9 +4,10 @@ import { join } from 'node:path' import { afterEach, describe, expect, it } from 'vitest' import Database from '../../sqlite/sync-database' import { LEGACY_RUN_ID, OrchestrationDb } from './db' -import type { MessageType } from './db' import { createRootDispatch } from './db/root-dispatch-test-fixture' +const runId = 'run_legacy_local' + // Overwrites the datetime('now')-seeded timestamps with explicit fixture values // so stale-detection assertions stay deterministic (no wall clock). function setDispatchTimes( @@ -33,154 +34,10 @@ describe('OrchestrationDb', () => { return db } - describe('messages', () => { - it('inserts and retrieves a message', () => { - const d = createDb() - const msg = d.insertMessage({ - from: 'term_a', - to: 'term_b', - subject: 'hello', - body: 'world' - }) - expect(msg.id).toMatch(/^msg_/) - expect(msg.from_handle).toBe('term_a') - expect(msg.to_handle).toBe('term_b') - expect(msg.subject).toBe('hello') - expect(msg.body).toBe('world') - expect(msg.type).toBe('status') - expect(msg.priority).toBe('normal') - expect(msg.read).toBe(0) - expect(msg.sequence).toBeGreaterThan(0) - }) - - it('returns unread messages in sequence order', () => { - const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 'first' }) - d.insertMessage({ from: 'a', to: 'b', subject: 'second' }) - d.insertMessage({ from: 'a', to: 'c', subject: 'other' }) - - const unread = d.getUnreadMessages('b') - expect(unread).toHaveLength(2) - expect(unread[0].subject).toBe('first') - expect(unread[1].subject).toBe('second') - }) - - it('filters unread by type', () => { - const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 'status msg', type: 'status' }) - d.insertMessage({ from: 'a', to: 'b', subject: 'done msg', type: 'worker_done' }) - - const filtered = d.getUnreadMessages('b', ['worker_done']) - expect(filtered).toHaveLength(1) - expect(filtered[0].type).toBe('worker_done') - }) - - it('excludes already-delivered rows from getUndeliveredUnreadMessages', () => { - const d = createDb() - const m1 = d.insertMessage({ from: 'a', to: 'b', subject: 'one' }) - const m2 = d.insertMessage({ from: 'a', to: 'b', subject: 'two' }) - - d.markAsDelivered([m1.id]) - - // Push delivery query: only undelivered, unread. - const pending = d.getUndeliveredUnreadMessages('b') - expect(pending).toHaveLength(1) - expect(pending[0].id).toBe(m2.id) - - // Explicit `check` still sees both (they are still unread). - const unread = d.getUnreadMessages('b') - expect(unread).toHaveLength(2) - }) - - it('creates the undelivered inbox index used by push delivery', () => { - const d = createDb() - const sqlite = (d as unknown as { db: Database.Database }).db - - const indexes = sqlite - .prepare( - `SELECT name FROM sqlite_master WHERE type = 'index' AND tbl_name = 'messages' AND name = 'idx_messages_undelivered_inbox'` - ) - .all() - - expect(indexes).toHaveLength(1) - }) - - it('filters getUndeliveredUnreadMessages by type', () => { - const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 's', type: 'status' }) - const wd = d.insertMessage({ from: 'a', to: 'b', subject: 'd', type: 'worker_done' }) - - const filtered = d.getUndeliveredUnreadMessages('b', ['worker_done']) - expect(filtered).toHaveLength(1) - expect(filtered[0].id).toBe(wd.id) - }) - - it('marks messages as read', () => { - const d = createDb() - const m1 = d.insertMessage({ from: 'a', to: 'b', subject: 'one' }) - const m2 = d.insertMessage({ from: 'a', to: 'b', subject: 'two' }) - - d.markAsRead([m1.id]) - - const unread = d.getUnreadMessages('b') - expect(unread).toHaveLength(1) - expect(unread[0].id).toBe(m2.id) - }) - - it('stores typed payload and thread_id', () => { - const d = createDb() - const payload = JSON.stringify({ taskId: 'task_abc', filesModified: ['src/a.ts'] }) - const msg = d.insertMessage({ - from: 'a', - to: 'b', - subject: 'done', - type: 'worker_done', - priority: 'high', - threadId: 'thread_1', - payload - }) - - expect(msg.type).toBe('worker_done') - expect(msg.priority).toBe('high') - expect(msg.thread_id).toBe('thread_1') - expect(msg.payload).toBe(payload) - }) - - it('rejects invalid message type', () => { - const d = createDb() - expect(() => - d.insertMessage({ - from: 'a', - to: 'b', - subject: 'bad', - type: 'invalid' as MessageType - }) - ).toThrow() - }) - - it('getInbox returns all messages across recipients', () => { - const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 'one' }) - d.insertMessage({ from: 'a', to: 'c', subject: 'two' }) - d.insertMessage({ from: 'b', to: 'a', subject: 'three' }) - - const inbox = d.getInbox(10) - expect(inbox).toHaveLength(3) - }) - - it('getMessageById returns the correct message', () => { - const d = createDb() - const msg = d.insertMessage({ from: 'a', to: 'b', subject: 'test' }) - const found = d.getMessageById(msg.id) - expect(found?.subject).toBe('test') - expect(d.getMessageById('msg_nonexistent')).toBeUndefined() - }) - }) - describe('tasks', () => { it('creates a task with no deps as ready', () => { const d = createDb() - const task = d.createTask({ spec: 'do something' }) + const task = d.createTask({ runId, spec: 'do something' }) expect(task.id).toMatch(/^task_/) expect(task.status).toBe('ready') expect(task.deps).toBe('[]') @@ -191,6 +48,7 @@ describe('OrchestrationDb', () => { it('persists explicit task display metadata', () => { const d = createDb() const task = d.createTask({ + runId, spec: 'full details', taskTitle: 'Checkout race', displayName: 'Fix checkout race' @@ -204,6 +62,7 @@ describe('OrchestrationDb', () => { it('persists the creating terminal handle for task-created worktrees', () => { const d = createDb() const task = d.createTask({ + runId, spec: 'spawn related workspace', createdByTerminalHandle: 'term_creator' }) @@ -214,16 +73,16 @@ describe('OrchestrationDb', () => { it('creates a task with deps as pending', () => { const d = createDb() - const parent = d.createTask({ spec: 'parent' }) - const child = d.createTask({ spec: 'child', deps: [parent.id] }) + const parent = d.createTask({ runId, spec: 'parent' }) + const child = d.createTask({ runId, spec: 'child', deps: [parent.id] }) expect(child.status).toBe('pending') expect(JSON.parse(child.deps)).toEqual([parent.id]) }) it('promotes pending tasks when deps complete', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second', deps: [t1.id] }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second', deps: [t1.id] }) expect(d.getTask(t2.id)?.status).toBe('pending') @@ -234,9 +93,9 @@ describe('OrchestrationDb', () => { it('does not promote task until ALL deps complete', () => { const d = createDb() - const t1 = d.createTask({ spec: 'a' }) - const t2 = d.createTask({ spec: 'b' }) - const t3 = d.createTask({ spec: 'c', deps: [t1.id, t2.id] }) + const t1 = d.createTask({ runId, spec: 'a' }) + const t2 = d.createTask({ runId, spec: 'b' }) + const t3 = d.createTask({ runId, spec: 'c', deps: [t1.id, t2.id] }) d.updateTaskStatus(t1.id, 'completed') expect(d.getTask(t3.id)?.status).toBe('pending') @@ -247,7 +106,7 @@ describe('OrchestrationDb', () => { it('sets completed_at on completion', () => { const d = createDb() - const task = d.createTask({ spec: 'do it' }) + const task = d.createTask({ runId, spec: 'do it' }) const updated = d.updateTaskStatus(task.id, 'completed', '{"result": true}') expect(updated?.completed_at).toBeTruthy() expect(updated?.result).toBe('{"result": true}') @@ -255,7 +114,7 @@ describe('OrchestrationDb', () => { it('completing a task frees its active dispatch context', () => { const d = createDb() - const task = d.createTask({ spec: 'do it' }) + const task = d.createTask({ runId, spec: 'do it' }) createRootDispatch(d, task.id, 'term_a') d.updateTaskStatus(task.id, 'completed') @@ -266,8 +125,8 @@ describe('OrchestrationDb', () => { it('listTasks filters by status', () => { const d = createDb() - d.createTask({ spec: 'ready task' }) - const t2 = d.createTask({ spec: 'another' }) + d.createTask({ runId, spec: 'ready task' }) + const t2 = d.createTask({ runId, spec: 'another' }) d.updateTaskStatus(t2.id, 'completed') expect(d.listTasks({ status: 'ready' })).toHaveLength(1) @@ -277,15 +136,15 @@ describe('OrchestrationDb', () => { it('listTasks returns all when no filter', () => { const d = createDb() - d.createTask({ spec: 'one' }) - d.createTask({ spec: 'two' }) + d.createTask({ runId, spec: 'one' }) + d.createTask({ runId, spec: 'two' }) expect(d.listTasks()).toHaveLength(2) }) it('listTasksWithDispatch joins active dispatch metadata', () => { const d = createDb() - const ready = d.createTask({ spec: 'ready task' }) - const dispatched = d.createTask({ spec: 'active task' }) + const ready = d.createTask({ runId, spec: 'ready task' }) + const dispatched = d.createTask({ runId, spec: 'active task' }) const ctx = createRootDispatch(d, dispatched.id, 'term_worker') const rows = d.listTasksWithDispatch() @@ -300,7 +159,7 @@ describe('OrchestrationDb', () => { it('listTasksWithDispatch does not surface completed dispatches', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) createRootDispatch(d, task.id, 'term_worker') d.updateTaskStatus(task.id, 'completed') @@ -314,8 +173,8 @@ describe('OrchestrationDb', () => { it('supports parent_id for task decomposition', () => { const d = createDb() - const parent = d.createTask({ spec: 'parent' }) - const child = d.createTask({ spec: 'child', parentId: parent.id }) + const parent = d.createTask({ runId, spec: 'parent' }) + const child = d.createTask({ runId, spec: 'child', parentId: parent.id }) expect(child.parent_id).toBe(parent.id) }) }) @@ -323,7 +182,7 @@ describe('OrchestrationDb', () => { describe('dispatch contexts', () => { it('creates a dispatch context and marks task as dispatched', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(d, task.id, 'term_worker') expect(ctx.id).toMatch(/^ctx_/) @@ -335,8 +194,8 @@ describe('OrchestrationDb', () => { it('rejects dispatch for non-ready tasks', () => { const d = createDb() - const parent = d.createTask({ spec: 'parent' }) - const child = d.createTask({ spec: 'child', deps: [parent.id] }) + const parent = d.createTask({ runId, spec: 'parent' }) + const child = d.createTask({ runId, spec: 'child', deps: [parent.id] }) expect(() => createRootDispatch(d, child.id, 'term_worker')).toThrow( /only ready tasks can be dispatched/ @@ -345,8 +204,8 @@ describe('OrchestrationDb', () => { it('rejects dispatch to an occupied terminal', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second' }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second' }) createRootDispatch(d, t1.id, 'term_worker') expect(() => createRootDispatch(d, t2.id, 'term_worker')).toThrow( @@ -361,8 +220,8 @@ describe('OrchestrationDb', () => { it('rejects dispatch to a reminted handle on a pane with an active dispatch', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second' }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second' }) createRootDispatch(d, t1.id, 'term_old', `tab_1:${LEAF_A}`) expect(() => createRootDispatch(d, t2.id, 'term_new', `tab_1:${LEAF_A}`)).toThrow( @@ -372,8 +231,8 @@ describe('OrchestrationDb', () => { it('rejects dispatch when pane keys share a leaf after break-out', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second' }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second' }) createRootDispatch(d, t1.id, 'term_old', `tab_1:${LEAF_A}`) expect(() => createRootDispatch(d, t2.id, 'term_new', `tab_2:${LEAF_A}`)).toThrow( @@ -383,8 +242,8 @@ describe('OrchestrationDb', () => { it('allows concurrent dispatches to different panes', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second' }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second' }) createRootDispatch(d, t1.id, 'term_a', `tab_1:${LEAF_A}`) expect(() => createRootDispatch(d, t2.id, 'term_b', `tab_1:${LEAF_B}`)).not.toThrow() @@ -392,8 +251,8 @@ describe('OrchestrationDb', () => { it('falls back to handle lock when pane keys are missing', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second' }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second' }) createRootDispatch(d, t1.id, 'term_worker') // New dispatch has a pane key but the active row is legacy (no pane key): @@ -403,8 +262,8 @@ describe('OrchestrationDb', () => { it('allows dispatch to a terminal after previous dispatch completes', () => { const d = createDb() - const t1 = d.createTask({ spec: 'first' }) - const t2 = d.createTask({ spec: 'second' }) + const t1 = d.createTask({ runId, spec: 'first' }) + const t2 = d.createTask({ runId, spec: 'second' }) const ctx1 = createRootDispatch(d, t1.id, 'term_worker') d.completeDispatch(ctx1.id) @@ -414,7 +273,7 @@ describe('OrchestrationDb', () => { it('getDispatchContext returns latest for a task', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(d, task.id, 'term_a') const found = d.getDispatchContext(task.id) expect(found?.id).toBe(ctx.id) @@ -422,7 +281,7 @@ describe('OrchestrationDb', () => { it('getDispatchContext uses insertion order when timestamps tie', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx1 = createRootDispatch(d, task.id, 'term_a') d.failDispatch(ctx1.id, 'retry') const ctx2 = createRootDispatch(d, task.id, 'term_a') @@ -432,7 +291,7 @@ describe('OrchestrationDb', () => { it('getActiveDispatchForTerminal returns active dispatch', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) createRootDispatch(d, task.id, 'term_a') const active = d.getActiveDispatchForTerminal('term_a') @@ -442,10 +301,16 @@ describe('OrchestrationDb', () => { it('getLatestDispatchForTerminal returns the most recent completed dispatch', () => { const d = createDb() - const firstTask = d.createTask({ spec: 'first' }) + const firstTask = d.createTask({ + runId, + spec: 'first' + }) const first = createRootDispatch(d, firstTask.id, 'term_a') d.completeDispatch(first.id) - const secondTask = d.createTask({ spec: 'second' }) + const secondTask = d.createTask({ + runId, + spec: 'second' + }) const second = createRootDispatch(d, secondTask.id, 'term_a') d.completeDispatch(second.id) @@ -457,7 +322,7 @@ describe('OrchestrationDb', () => { it('circuit breaker trips after 3 failures', () => { const d = createDb() - const task = d.createTask({ spec: 'flaky' }) + const task = d.createTask({ runId, spec: 'flaky' }) const ctx = createRootDispatch(d, task.id, 'term_a') const after1 = d.failDispatch(ctx.id, 'timeout') @@ -480,7 +345,7 @@ describe('OrchestrationDb', () => { it('completeDispatch sets completed_at', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(d, task.id, 'term_a') d.completeDispatch(ctx.id) @@ -493,7 +358,10 @@ describe('OrchestrationDb', () => { describe('decision gates', () => { it('creates a gate and blocks the task', () => { const d = createDb() - const task = d.createTask({ spec: 'needs approval' }) + const task = d.createTask({ + runId, + spec: 'needs approval' + }) createRootDispatch(d, task.id, 'term_a') const gate = d.createGate({ taskId: task.id, @@ -513,7 +381,7 @@ describe('OrchestrationDb', () => { it('resolves a gate and unblocks the task', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const gate = d.createGate({ taskId: task.id, question: 'ok?' }) const resolved = d.resolveGate(gate.id, 'yes') @@ -526,7 +394,7 @@ describe('OrchestrationDb', () => { it('times out a gate', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const gate = d.createGate({ taskId: task.id, question: 'ok?' }) const timedOut = d.timeoutGate(gate.id) @@ -535,8 +403,8 @@ describe('OrchestrationDb', () => { it('lists gates with filters', () => { const d = createDb() - const t1 = d.createTask({ spec: 'a' }) - const t2 = d.createTask({ spec: 'b' }) + const t1 = d.createTask({ runId, spec: 'a' }) + const t2 = d.createTask({ runId, spec: 'b' }) d.createGate({ taskId: t1.id, question: 'q1' }) const g2 = d.createGate({ taskId: t2.id, question: 'q2' }) d.resolveGate(g2.id, 'done') @@ -599,8 +467,13 @@ describe('OrchestrationDb', () => { describe('lifecycle', () => { it('resetAll clears all tables', () => { const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 'test' }) - d.createTask({ spec: 'work' }) + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'test' + }) + d.createTask({ runId, spec: 'work' }) d.resetAll() @@ -610,8 +483,13 @@ describe('OrchestrationDb', () => { it('resetMessages clears only messages', () => { const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 'test' }) - d.createTask({ spec: 'work' }) + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'test' + }) + d.createTask({ runId, spec: 'work' }) d.resetMessages() @@ -621,8 +499,13 @@ describe('OrchestrationDb', () => { it('resetTasks clears tasks and dispatch contexts', () => { const d = createDb() - d.insertMessage({ from: 'a', to: 'b', subject: 'test' }) - const task = d.createTask({ spec: 'work' }) + d.insertMessage({ + runId, + from: 'a', + to: 'b', + subject: 'test' + }) + const task = d.createTask({ runId, spec: 'work' }) createRootDispatch(d, task.id, 'term_a') d.resetTasks() @@ -636,6 +519,7 @@ describe('OrchestrationDb', () => { it('insertMessage accepts type = heartbeat', () => { const d = createDb() const msg = d.insertMessage({ + runId, from: 'worker', to: 'coord', subject: 'alive', @@ -647,7 +531,7 @@ describe('OrchestrationDb', () => { it('recordHeartbeat updates last_heartbeat_at on dispatched rows', () => { const d = createDb() - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(d, task.id, 'term_a') d.recordHeartbeat(ctx.id, '2026-05-04T00:00:00.000Z') @@ -662,10 +546,10 @@ describe('OrchestrationDb', () => { // (b) dispatched, heartbeated 12 min ago → STALE (expected result) // (c) dispatched, never heartbeated, dispatched 30s ago → not stale (grace) // (d) completed, heartbeated 30 min ago → not stale (status filter) - const taskA = d.createTask({ spec: 'a' }) - const taskB = d.createTask({ spec: 'b' }) - const taskC = d.createTask({ spec: 'c' }) - const taskD = d.createTask({ spec: 'd' }) + const taskA = d.createTask({ runId, spec: 'a' }) + const taskB = d.createTask({ runId, spec: 'b' }) + const taskC = d.createTask({ runId, spec: 'c' }) + const taskD = d.createTask({ runId, spec: 'd' }) const ctxA = createRootDispatch(d, taskA.id, 'term_a') const ctxB = createRootDispatch(d, taskB.id, 'term_b') const ctxC = createRootDispatch(d, taskC.id, 'term_c') @@ -710,15 +594,19 @@ describe('OrchestrationDb', () => { // Fresh worker: dispatched 12:00, heartbeat 12:05 (space-format), both // after the 11:55 threshold → NOT stale. - const fresh = createRootDispatch(d, d.createTask({ spec: 'fresh' }).id, 'term_fresh') + const fresh = createRootDispatch(d, d.createTask({ runId, spec: 'fresh' }).id, 'term_fresh') setDispatchTimes(d, fresh.id, '2026-07-12 12:00:00', '2026-07-12 12:05:00') // Legacy ISO-format fresh row (mixed-format table) stays fresh too. - const legacy = createRootDispatch(d, d.createTask({ spec: 'legacy' }).id, 'term_legacy') + const legacy = createRootDispatch( + d, + d.createTask({ runId, spec: 'legacy' }).id, + 'term_legacy' + ) setDispatchTimes(d, legacy.id, '2026-07-12T12:00:00.000Z', '2026-07-12T12:05:00.000Z') // Genuinely hung: dispatched + heartbeated at 10:00, ~2h before threshold. - const hung = createRootDispatch(d, d.createTask({ spec: 'hung' }).id, 'term_hung') + const hung = createRootDispatch(d, d.createTask({ runId, spec: 'hung' }).id, 'term_hung') setDispatchTimes(d, hung.id, '2026-07-12 10:00:00', '2026-07-12 10:00:00') const stale = d.getStaleDispatches('2026-07-12T11:55:00.000Z') @@ -730,7 +618,7 @@ describe('OrchestrationDb', () => { // Space-format dispatched_at one minute after the threshold, no heartbeat // yet → still inside the grace window, must not be flagged. - const ctx = createRootDispatch(d, d.createTask({ spec: 'x' }).id, 'term_x') + const ctx = createRootDispatch(d, d.createTask({ runId, spec: 'x' }).id, 'term_x') setDispatchTimes(d, ctx.id, '2026-07-12 12:00:00') const stale = d.getStaleDispatches('2026-07-12T11:59:00.000Z') @@ -742,7 +630,11 @@ describe('OrchestrationDb', () => { it('getStaleDispatches keeps a fresh row just after a UTC-midnight threshold (#8452)', () => { const d = createDb() - const ctx = createRootDispatch(d, d.createTask({ spec: 'midnight' }).id, 'term_midnight') + const ctx = createRootDispatch( + d, + d.createTask({ runId, spec: 'midnight' }).id, + 'term_midnight' + ) setDispatchTimes(d, ctx.id, '2026-05-04 00:04:00') const stale = d.getStaleDispatches('2026-05-04T00:00:00.000Z') @@ -755,7 +647,7 @@ describe('OrchestrationDb', () => { it('getStaleDispatches keeps a live worker with a fresh space-format heartbeat (#8452)', () => { const d = createDb() - const ctx = createRootDispatch(d, d.createTask({ spec: 'live' }).id, 'term_live') + const ctx = createRootDispatch(d, d.createTask({ runId, spec: 'live' }).id, 'term_live') setDispatchTimes(d, ctx.id, '2026-07-12 10:00:00', '2026-07-12 11:59:00') const stale = d.getStaleDispatches('2026-07-12T11:55:00.000Z') @@ -765,6 +657,7 @@ describe('OrchestrationDb', () => { it('getThreadMessagesFor returns only same-thread replies to a handle', () => { const d = createDb() const outbound = d.insertMessage({ + runId, from: 'worker', to: 'coord', subject: 'Question', @@ -773,6 +666,7 @@ describe('OrchestrationDb', () => { }) // Reply in the same thread addressed to the worker const reply = d.insertMessage({ + runId, from: 'coord', to: 'worker', subject: 'Re: Question', @@ -781,6 +675,7 @@ describe('OrchestrationDb', () => { }) // Distractor: different thread, same recipient d.insertMessage({ + runId, from: 'coord', to: 'worker', subject: 'other', @@ -789,6 +684,7 @@ describe('OrchestrationDb', () => { }) // Distractor: same thread but not addressed to worker d.insertMessage({ + runId, from: 'coord', to: 'someone_else', subject: 'cc', @@ -902,6 +798,7 @@ describe('OrchestrationDb', () => { // (a) INSERT type='heartbeat' now succeeds expect(() => d.insertMessage({ + runId, from: 'w', to: 'c', subject: 'alive', @@ -911,7 +808,7 @@ describe('OrchestrationDb', () => { ).not.toThrow() // (b) last_heartbeat_at column exists on dispatch_contexts - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(d, task.id, 'term_a') d.recordHeartbeat(ctx.id, '2026-05-04T00:00:00.000Z') expect(d.getDispatchContext(task.id)?.last_heartbeat_at).toBe('2026-05-04T00:00:00.000Z') @@ -942,11 +839,12 @@ describe('OrchestrationDb', () => { const d = new OrchestrationDb(path) db = d - const task = d.createTask({ spec: 'work' }) + const task = d.createTask({ runId, spec: 'work' }) const ctx = createRootDispatch(d, task.id, 'term_a', 'tab_1:leaf_1') expect(d.getDispatchContextById(ctx.id)?.assignee_pane_key).toBe('tab_1:leaf_1') const msg = d.insertMessage({ + runId, from: 'w', to: 'c', subject: 'done', @@ -960,6 +858,7 @@ describe('OrchestrationDb', () => { const path = createV1Snapshot() const first = new OrchestrationDb(path) first.insertMessage({ + runId, from: 'w', to: 'c', subject: 'alive', @@ -972,6 +871,7 @@ describe('OrchestrationDb', () => { db = second expect(() => second.insertMessage({ + runId, from: 'w', to: 'c', subject: 'again', diff --git a/src/main/runtime/orchestration/db/attempt-outcome-projection.test.ts b/src/main/runtime/orchestration/db/attempt-outcome-projection.test.ts index eae300a7b20..a8dc098728a 100644 --- a/src/main/runtime/orchestration/db/attempt-outcome-projection.test.ts +++ b/src/main/runtime/orchestration/db/attempt-outcome-projection.test.ts @@ -48,7 +48,7 @@ describe('durable Attempt observation and outcome projection', () => { function createAttempt(): { taskId: string; dispatchId: string } { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'observe outcome' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'observe outcome' }) const dispatch = createRootDispatch(db, task.id, 'term_observed') return { taskId: task.id, dispatchId: dispatch.id } } @@ -162,7 +162,10 @@ describe('durable Attempt observation and outcome projection', () => { const path = join(dir, 'orchestration.sqlite') try { db = new OrchestrationDb(path) - const task = db.createTask({ spec: 'durable observation' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'durable observation' + }) const dispatch = createRootDispatch(db, task.id, 'term_durable') db.recordAttemptObservation( fact(dispatch.id, { @@ -189,7 +192,10 @@ describe('durable Attempt observation and outcome projection', () => { it('keeps worker_done settlement as the atomic success fast path', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'worker_done fast path' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'worker_done fast path' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/db/contract-constants.ts b/src/main/runtime/orchestration/db/contract-constants.ts index 287ce565d5b..56c2e2d542f 100644 --- a/src/main/runtime/orchestration/db/contract-constants.ts +++ b/src/main/runtime/orchestration/db/contract-constants.ts @@ -1,10 +1,20 @@ -import { ORCHESTRATION_LEGACY_RUN_ID } from '../../../../shared/orchestration-rpc-contract' +import { + ORCHESTRATION_LEGACY_RUN_ID, + ORCHESTRATION_UNBOUND_RUN_ID +} from '../../../../shared/orchestration-rpc-contract' import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../shared/protocol-version' export const LEGACY_RUN_ID = ORCHESTRATION_LEGACY_RUN_ID +export const UNBOUND_RUN_ID = ORCHESTRATION_UNBOUND_RUN_ID + +// Why: a v1.4.198 coordinator sends no Run id, so its remote workers file mail under a per-attachment stub Run. +export const FEDERATED_STUB_HOME_RUN_ID_PREFIX = 'run_federated_' +export function federatedStubHomeRunId(dispatchId: string): string { + return `${FEDERATED_STUB_HOME_RUN_ID_PREFIX}${dispatchId}` +} export const LEGACY_CONTRACT_VERSION = 0 export const CURRENT_CONTRACT_VERSION = ORCHESTRATION_CONTRACT_VERSION // Schema versions: v2 'heartbeat'+last_heartbeat_at, v3 delivered_at, v4 task-creator terminal, v5 task_title/display_name, v6 pane identity, v7 lightweight Runs, v8 crash-safe Run deliveries, v9 durable question threads, v10 Dispatch capabilities, v11 durable mutation receipts, v12 composed worker state, v18 post-v6 version-skew repair, v19 adopted legacy Runs and compatibility receipts, v20 legacy question backfill, v21 legacy scheduler-loss provenance, v22 dispatch assignee lookup, v23 worker terminal resource ownership, v24 creator-incarnation authority, v25 active Dispatch handle lookup, v26 indexed mutation receipt capacity, v27 durable federation acknowledgments, v28 durable local mutation caller identity, v31 dispatch/resource identity links, v32 bounded worker-terminal recovery metadata, v33 durable mailbox pointer Enter state, v34 role-addressed mailbox deliveries, v35 mailbox delivery default and index-predicate repair, v36 dispatch mailbox consumer generation, v37 recorded dispatch creator identity, v39 structured session journal archives. -export const SCHEMA_VERSION = 39 +export const SCHEMA_VERSION = 40 diff --git a/src/main/runtime/orchestration/db/decision-gate-lifecycle.test.ts b/src/main/runtime/orchestration/db/decision-gate-lifecycle.test.ts index 219cf6fe212..9fdc9d9b5fb 100644 --- a/src/main/runtime/orchestration/db/decision-gate-lifecycle.test.ts +++ b/src/main/runtime/orchestration/db/decision-gate-lifecycle.test.ts @@ -9,7 +9,7 @@ describe('decision-gate lifecycle transitions', () => { it('blocks the dispatched Task when creating a gate', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'gate blocks task' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'gate blocks task' }) createRootDispatch(db, task.id, 'term_gate') expect(db.getTask(task.id)?.status).toBe('dispatched') @@ -20,7 +20,7 @@ describe('decision-gate lifecycle transitions', () => { it('rolls back the gate row when the Task transition cannot commit', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'atomic gate creation' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'atomic gate creation' }) const dispatch = createRootDispatch(db, task.id, 'term_gate') db.db.exec(` CREATE TRIGGER reject_gate_task_block diff --git a/src/main/runtime/orchestration/db/decision-gates/decision-gate-store.ts b/src/main/runtime/orchestration/db/decision-gates/decision-gate-store.ts index 532fa52b22a..296bcea42bc 100644 --- a/src/main/runtime/orchestration/db/decision-gates/decision-gate-store.ts +++ b/src/main/runtime/orchestration/db/decision-gates/decision-gate-store.ts @@ -1,6 +1,5 @@ import type { DecisionGateRow, DispatchContextRow, GateStatus } from '../../types' import { OrchestrationError } from '../../orchestration-error' -import { LEGACY_RUN_ID } from '../contract-constants' import { generateId } from '../generated-id' import type { OrchestrationDb } from '../orchestration-db' import { transitionLifecycleWithDb } from '../lifecycle-transition' @@ -18,6 +17,16 @@ export function createGate( ): DecisionGateRow { this.db.exec('SAVEPOINT create_gate') try { + const task = this.getTask(gate.taskId) + if (!task) { + throw new OrchestrationError( + 'lifecycle_not_found', + `Task ${gate.taskId} was not found while creating a decision gate.`, + { taskId: gate.taskId } + ) + } + const runId = task.run_id + this.requireRun(runId) const active = this.db .prepare( `SELECT * FROM dispatch_contexts @@ -65,22 +74,8 @@ export function createGate( .prepare( 'INSERT INTO decision_gates (id, run_id, task_id, question, options) VALUES (?, ?, ?, ?, ?)' ) - .run( - id, - this.getTask(gate.taskId)?.run_id ?? LEGACY_RUN_ID, - gate.taskId, - gate.question, - optionsJson - ) + .run(id, runId, gate.taskId, gate.question, optionsJson) this.completeActiveDispatchesForTask(gate.taskId) - const task = this.getTask(gate.taskId) - if (!task) { - throw new OrchestrationError( - 'lifecycle_not_found', - `Task ${gate.taskId} was not found while creating a decision gate.`, - { taskId: gate.taskId } - ) - } transitionLifecycleWithDb(this.db, { entity: 'task', id: gate.taskId, diff --git a/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts b/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts index 3584a2e59a6..ddb0e383143 100644 --- a/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts +++ b/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts @@ -109,9 +109,13 @@ export function settleWorkerReportInTransaction( (dispatch.status === 'pending' || dispatch.status === 'dispatched') && task.status === 'blocked' && reportingWorker?.state === 'start_unknown' + const reportingStart = + dispatch.status === 'pending' && + task.status === 'dispatched' && + reportingWorker?.state === 'starting' const previousDispatchStatus = settledByUnobservedPrompt ? 'failed' - : reconnectingStart + : reconnectingStart || reportingStart ? dispatch.status : 'dispatched' const previousTaskStatus = settledByUnobservedPrompt @@ -198,7 +202,7 @@ export function settleWorkerReportInTransaction( const dispatchTransition = transitionLifecycleWithDb(this.db, { entity: 'dispatch', id: params.dispatchId, - from: reconnectingStart ? ['pending', 'dispatched'] : 'dispatched', + from: reconnectingStart || reportingStart ? ['pending', 'dispatched'] : 'dispatched', to: expectedDispatchStatus, projection: { completed_at: new Date().toISOString(), @@ -234,11 +238,11 @@ export function settleWorkerReportInTransaction( projection: { stage: 'settled', updated_at: new Date().toISOString() }, correction: 'unobserved_prompt_report' }) - } else if (reconnectingStart && params.outcome === 'succeeded') { + } else if ((reconnectingStart || reportingStart) && params.outcome === 'succeeded') { transitionLifecycleWithDb(this.db, { entity: 'worker', id: params.dispatchId, - from: 'start_unknown', + from: reportingStart ? 'starting' : 'start_unknown', to: 'ready' }) transitionLifecycleWithDb(this.db, { @@ -253,7 +257,7 @@ export function settleWorkerReportInTransaction( entity: 'worker', id: params.dispatchId, // A start_unknown success report reconnects through 'ready' above; only failure settles here. - from: params.outcome === 'succeeded' ? 'ready' : ['ready', 'start_unknown'], + from: params.outcome === 'succeeded' ? 'ready' : ['ready', 'start_unknown', 'starting'], to: params.outcome === 'succeeded' ? 'succeeded' : 'failed', projection: { stage: 'settled', updated_at: new Date().toISOString() } }) diff --git a/src/main/runtime/orchestration/db/dispatch-depth.test.ts b/src/main/runtime/orchestration/db/dispatch-depth.test.ts index 97df3f01c51..013cedffe91 100644 --- a/src/main/runtime/orchestration/db/dispatch-depth.test.ts +++ b/src/main/runtime/orchestration/db/dispatch-depth.test.ts @@ -16,7 +16,7 @@ describe('nested worker depth', () => { function coordinatorDispatchesWorker(maxDepth = UNCAPPED) { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'root task' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'root task' }) const worker = db.createDispatchContext({ taskId: task.id, assigneeHandle: 'term_worker', @@ -33,7 +33,7 @@ describe('nested worker depth', () => { it('refuses a worker dispatching a sub-worker at the default cap', () => { coordinatorDispatchesWorker() - const nested = db.createTask({ spec: 'nested task' }) + const nested = db.createTask({ runId: 'run_legacy_local', spec: 'nested task' }) expect(() => db.createDispatchContext({ taskId: nested.id, @@ -51,7 +51,7 @@ describe('nested worker depth', () => { it('tells the refused worker to complete the task itself', () => { coordinatorDispatchesWorker() - const nested = db.createTask({ spec: 'nested task' }) + const nested = db.createTask({ runId: 'run_legacy_local', spec: 'nested task' }) expect(() => db.createDispatchContext({ taskId: nested.id, @@ -64,7 +64,7 @@ describe('nested worker depth', () => { it('permits one more generation when the cap is raised, and records depth 2', () => { coordinatorDispatchesWorker() - const nested = db.createTask({ spec: 'nested task' }) + const nested = db.createTask({ runId: 'run_legacy_local', spec: 'nested task' }) const sub = db.createDispatchContext({ taskId: nested.id, assigneeHandle: 'term_sub', @@ -113,9 +113,9 @@ describe('nested worker depth', () => { db.db .prepare( `INSERT INTO remote_dispatch_attachments - (dispatch_id, task_id, home_peer_fingerprint, protocol_version, runtime_epoch, + (dispatch_id, task_id, home_run_id, home_peer_fingerprint, protocol_version, runtime_epoch, pane_key, process_incarnation, state, depth) - VALUES (?, ?, 'peer', 1, 'epoch', ?, ?, ?, ?)` + VALUES (?, ?, 'run_home', 'peer', 1, 'epoch', ?, ?, ?, ?)` ) .run(`ctx_${state}_${depth}_${paneKey}_${inc}`, 'task_remote', paneKey, inc, state, depth) } @@ -182,7 +182,10 @@ describe('nested worker depth', () => { it('takes the maximum when a process holds both a local and a remote role', () => { // Query order must not decide the answer: the deeper role governs. db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'local role' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'local role' + }) db.createDispatchContext({ taskId: task.id, assigneeHandle: 'term_both', @@ -220,13 +223,19 @@ describe('nested worker depth', () => { it('stamps depth 1 for a root coordinator', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'root work' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'root work' + }) expect(startWorker(task.id, SYSTEM, UNCAPPED).dispatch.depth).toBe(1) }) it('refuses a worker starting a sub-worker at the default cap', () => { coordinatorDispatchesWorker() - const nested = db.createTask({ spec: 'nested work' }) + const nested = db.createTask({ + runId: 'run_legacy_local', + spec: 'nested work' + }) expect(() => startWorker( nested.id, @@ -238,7 +247,10 @@ describe('nested worker depth', () => { it('refuses a worker retrying into a sub-worker at the default cap', () => { coordinatorDispatchesWorker() - const nested = db.createTask({ spec: 'nested retry work' }) + const nested = db.createTask({ + runId: 'run_legacy_local', + spec: 'nested retry work' + }) const first = startWorker(nested.id, SYSTEM, UNCAPPED) db.failWorkerStart(first.dispatch.id, 'accepted', 'first attempt failed') expect(() => @@ -257,7 +269,10 @@ describe('nested worker depth', () => { // Context-only dispatch stores null on purpose; requiring an incarnation // locally would silently drop real parents and fail open. db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'context only' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'context only' + }) const row = db.createDispatchContext({ taskId: task.id, assigneeHandle: 'term_ctx', diff --git a/src/main/runtime/orchestration/db/dispatch-mailbox-consumer-fencing.test.ts b/src/main/runtime/orchestration/db/dispatch-mailbox-consumer-fencing.test.ts index c7d287e8bdb..29a2e468ee8 100644 --- a/src/main/runtime/orchestration/db/dispatch-mailbox-consumer-fencing.test.ts +++ b/src/main/runtime/orchestration/db/dispatch-mailbox-consumer-fencing.test.ts @@ -22,7 +22,7 @@ describe('dispatch mailbox consumer fencing', () => { afterEach(() => db.close()) function dispatchWithMail(subjects: string[]): { id: string; runId: string } { - const task = db.createTask({ spec: 'fenced worker work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'fenced worker work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker', PANE_A) for (const subject of subjects) { db.insertMessage({ @@ -116,7 +116,10 @@ describe('dispatch mailbox consumer fencing', () => { }) it('bumps and fences on the worker-start attach path', () => { - const task = db.createTask({ spec: 'worker-start attach' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'worker-start attach' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -149,6 +152,7 @@ describe('dispatch mailbox consumer fencing', () => { it('gives a federated attachment its own generation on the worker host', () => { const dispatchId = 'ctx_remote_fence' db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: 'task_remote', homePeerFingerprint: 'home-peer', @@ -187,7 +191,10 @@ describe('dispatch mailbox consumer fencing', () => { }) it('starts a retry Dispatch on a fresh mailbox address rather than sharing the old one', () => { - const task = db.createTask({ spec: 'work that fails once' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'work that fails once' + }) const first = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/db/dispatch-row-writer.ts b/src/main/runtime/orchestration/db/dispatch-row-writer.ts index 807814a87b1..84862ee343d 100644 --- a/src/main/runtime/orchestration/db/dispatch-row-writer.ts +++ b/src/main/runtime/orchestration/db/dispatch-row-writer.ts @@ -49,8 +49,8 @@ const STARTING_DISPATCH_CONTEXT_SQL = `INSERT INTO dispatch_contexts ( ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', datetime('now'))` const REMOTE_DISPATCH_ATTACHMENT_SQL = `INSERT INTO remote_dispatch_attachments ( - dispatch_id, task_id, home_peer_fingerprint, protocol_version, runtime_epoch, depth - ) VALUES (?, ?, ?, ?, ?, ?)` + dispatch_id, home_run_id, task_id, home_peer_fingerprint, protocol_version, runtime_epoch, depth + ) VALUES (?, ?, ?, ?, ?, ?, ?)` /** Last line of defence: a row that reached here unstamped would read as a root. */ function assertStampedDepth(depth: number): void { @@ -140,6 +140,7 @@ export function insertRemoteDispatchAttachmentRow( db: Database.Database, params: { dispatchId: string + runId: string taskId: string homePeerFingerprint: string protocolVersion: number @@ -151,6 +152,7 @@ export function insertRemoteDispatchAttachmentRow( assertStampedDepth(params.depth) db.prepare(REMOTE_DISPATCH_ATTACHMENT_SQL).run( params.dispatchId, + params.runId, params.taskId, params.homePeerFingerprint, params.protocolVersion, diff --git a/src/main/runtime/orchestration/db/federation/federated-dispatch-observation-fence.test.ts b/src/main/runtime/orchestration/db/federation/federated-dispatch-observation-fence.test.ts index e32bf27a00c..ecc3ca5e2c0 100644 --- a/src/main/runtime/orchestration/db/federation/federated-dispatch-observation-fence.test.ts +++ b/src/main/runtime/orchestration/db/federation/federated-dispatch-observation-fence.test.ts @@ -8,7 +8,10 @@ describe('federated Dispatch observation fence', () => { it('rejects out-of-order epochs and observations captured before release', () => { const database = (db = new OrchestrationDb(':memory:')) - const task = database.createTask({ spec: 'fenced federated observation' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'fenced federated observation' + }) const started = database.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/db/federation/federated-stub-home-run-backfill.ts b/src/main/runtime/orchestration/db/federation/federated-stub-home-run-backfill.ts new file mode 100644 index 00000000000..47c596e260d --- /dev/null +++ b/src/main/runtime/orchestration/db/federation/federated-stub-home-run-backfill.ts @@ -0,0 +1,16 @@ +import type Database from '../../../../sqlite/sync-database' +import { FEDERATED_STUB_HOME_RUN_ID_PREFIX } from '../contract-constants' + +// Why: a rolled-back v1.4.198 host inserts attachments with home_run_id='' after user_version is +// already 40, so this idempotent repair runs on every open, not only inside the v40 migration. +export function backfillFederatedStubHomeRuns(db: Database.Database): void { + db.exec(` + INSERT OR IGNORE INTO runs (id, objective, home_database, consumer_generation, legacy) + SELECT '${FEDERATED_STUB_HOME_RUN_ID_PREFIX}' || dispatch_id, + 'Coordinated from ' || home_peer_fingerprint, 'remote', 0, 0 + FROM remote_dispatch_attachments WHERE home_run_id = ''; + UPDATE remote_dispatch_attachments + SET home_run_id = '${FEDERATED_STUB_HOME_RUN_ID_PREFIX}' || dispatch_id + WHERE home_run_id = ''; + `) +} diff --git a/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-create.ts b/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-create.ts index 56d26ccfe3b..5a6e89b65ca 100644 --- a/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-create.ts +++ b/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-create.ts @@ -2,12 +2,15 @@ import type { WorkerDispatchState, RemoteDispatchAttachmentRow } from '../../typ import { OrchestrationError } from '../../orchestration-error' import { ensureMutationReceiptCapacity } from '../../mutation-receipt-capacity' import type { OrchestrationDb } from '../orchestration-db' +import { federatedStubHomeRunId } from '../contract-constants' import { insertRemoteDispatchAttachmentRow } from '../dispatch-row-writer' export function createRemoteDispatchAttachment( this: OrchestrationDb, params: { dispatchId: string + /** Absent from a v1.4.198 coordinator; replaced by a per-attachment stub Run. */ + runId?: string taskId: string homePeerFingerprint: string protocolVersion: number @@ -43,6 +46,17 @@ export function createRemoteDispatchAttachment( `Remote attachment request ${params.mutationReceipt.requestId} already exists.` ) } + const runId = params.runId ?? federatedStubHomeRunId(params.dispatchId) + if (!runId.trim()) { + throw new OrchestrationError('invalid_argument', 'Missing Run ID') + } + this.db + .prepare( + `INSERT OR IGNORE INTO runs (id, objective, home_database, consumer_generation, legacy) + VALUES (?, ?, 'remote', 0, 0)` + ) + .run(runId, `Coordinated from ${params.homePeerFingerprint}`) + this.requireRun(runId) ensureMutationReceiptCapacity(this.db) this.db .prepare( @@ -59,6 +73,7 @@ export function createRemoteDispatchAttachment( ) insertRemoteDispatchAttachmentRow(this.db, { dispatchId: params.dispatchId, + runId, taskId: params.taskId, homePeerFingerprint: params.homePeerFingerprint, protocolVersion: params.protocolVersion, diff --git a/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-release.test.ts b/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-release.test.ts index ab515ffb30c..0865d4b8972 100644 --- a/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-release.test.ts +++ b/src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-release.test.ts @@ -16,6 +16,7 @@ describe('the remote attachment release guard', () => { function settledAttachment(dispatchId: string): void { db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: `task_${dispatchId}`, homePeerFingerprint: 'home-peer', diff --git a/src/main/runtime/orchestration/db/lifecycle-transition.test.ts b/src/main/runtime/orchestration/db/lifecycle-transition.test.ts index eed4332879f..936208e77ef 100644 --- a/src/main/runtime/orchestration/db/lifecycle-transition.test.ts +++ b/src/main/runtime/orchestration/db/lifecycle-transition.test.ts @@ -8,7 +8,7 @@ describe('guarded lifecycle transitions', () => { it('rejects a stale prior state without changing the projection', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'guarded transition' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'guarded transition' }) expect(() => db!.transitionLifecycle({ @@ -23,7 +23,7 @@ describe('guarded lifecycle transitions', () => { it('composes its projection into the caller-owned transaction', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'caller-owned rollback' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'caller-owned rollback' }) db.db.exec('SAVEPOINT lifecycle_test') expect( @@ -49,7 +49,7 @@ describe('guarded lifecycle transitions', () => { ['completed', 'blocked'] ] as const)('preserves public task updates from %s to %s', (from, to) => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'manual status correction' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'manual status correction' }) db.db.prepare('UPDATE tasks SET status = ? WHERE id = ?').run(from, task.id) expect(db.updateTaskStatus(task.id, to)?.status).toBe(to) diff --git a/src/main/runtime/orchestration/db/messages/message-insert.ts b/src/main/runtime/orchestration/db/messages/message-insert.ts index 2984545a09b..51ce7aaaa50 100644 --- a/src/main/runtime/orchestration/db/messages/message-insert.ts +++ b/src/main/runtime/orchestration/db/messages/message-insert.ts @@ -1,9 +1,9 @@ import type { MessageType, MessagePriority, MessageDeliveryContract, MessageRow } from '../../types' -import { LEGACY_RUN_ID } from '../contract-constants' import { generateId } from '../generated-id' import { exposeMessageTimestamps } from '../utc-timestamp' import type { OrchestrationDb } from '../orchestration-db' import { runLifecycleWriteTransaction } from '../lifecycle-write-transaction-runner' +import { UNBOUND_RUN_ID } from '../contract-constants' // ── Messages ── @@ -26,7 +26,18 @@ export type MessageInsert = { } export function insertMessage(this: OrchestrationDb, msg: MessageInsert): MessageRow { - const runId = msg.runId ?? LEGACY_RUN_ID + // A sender in no Run (two plain terminals, `send --to `) still gets durable mail. It is + // filed under the unbound Run, never the legacy one, which the schema-skew probe reads as pre-Runs. + // Created on first use so `run list` shows it only to a user who has such mail. + const runId = msg.runId ?? UNBOUND_RUN_ID + if (msg.runId == null) { + this.db + .prepare( + `INSERT OR IGNORE INTO runs (id, objective, home_database, consumer_generation, legacy) + VALUES (?, 'Mail from terminals in no Run', 'this_database', 0, 0)` + ) + .run(UNBOUND_RUN_ID) + } const deliveryContract = msg.deliveryContract ?? 'current_delivery' this.requireRun(runId) const id = msg.id ?? generateId('msg') diff --git a/src/main/runtime/orchestration/db/orchestration-db.ts b/src/main/runtime/orchestration/db/orchestration-db.ts index 2a970841a38..1ce52e96c4a 100644 --- a/src/main/runtime/orchestration/db/orchestration-db.ts +++ b/src/main/runtime/orchestration/db/orchestration-db.ts @@ -1,6 +1,7 @@ import Database from '../../../sqlite/sync-database' import { attachOrchestrationDbMethods } from './attach-orchestration-db-methods' import { hardenOrchestrationDatabaseFiles } from './database-file-permissions' +import { backfillFederatedStubHomeRuns } from './federation/federated-stub-home-run-backfill' import type { OrchestrationDbMethods } from './orchestration-db-methods' import { createCoordinatorMailRoutingTrigger, @@ -28,6 +29,7 @@ class OrchestrationDbCore { this.db.pragma('busy_timeout = 5000') createTables.call(this as unknown as OrchestrationDb) migrate.call(this as unknown as OrchestrationDb) + backfillFederatedStubHomeRuns(this.db) createCoordinatorMailRoutingTrigger.call(this as unknown as OrchestrationDb) rememberCurrentRunCoordinatorHandles.call(this as unknown as OrchestrationDb) hardenOrchestrationDatabaseFiles(dbPath) diff --git a/src/main/runtime/orchestration/db/schema/create-graph-tables-sql.ts b/src/main/runtime/orchestration/db/schema/create-graph-tables-sql.ts index 5aed50eb7f9..d674a5298e7 100644 --- a/src/main/runtime/orchestration/db/schema/create-graph-tables-sql.ts +++ b/src/main/runtime/orchestration/db/schema/create-graph-tables-sql.ts @@ -38,6 +38,8 @@ CREATE TABLE IF NOT EXISTS federated_dispatches ( ); CREATE TABLE IF NOT EXISTS remote_dispatch_attachments ( + -- DEFAULT: a rolled-back v1.4.198 host still inserts here without a home Run. + home_run_id TEXT NOT NULL DEFAULT '', dispatch_id TEXT PRIMARY KEY, task_id TEXT NOT NULL, home_peer_fingerprint TEXT NOT NULL, diff --git a/src/main/runtime/orchestration/db/schema/federated-home-run-migration.test.ts b/src/main/runtime/orchestration/db/schema/federated-home-run-migration.test.ts new file mode 100644 index 00000000000..4fcd78634f3 --- /dev/null +++ b/src/main/runtime/orchestration/db/schema/federated-home-run-migration.test.ts @@ -0,0 +1,105 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../../shared/protocol-version' +import { OrchestrationDb } from '../orchestration-db' +import { SCHEMA_VERSION, federatedStubHomeRunId } from '../contract-constants' +import { migrateV40 } from './migrate-v40' +import { importFederatedControlMessage } from '../../federation-control-message' + +describe('federated home Run migration', () => { + let db: OrchestrationDb + beforeEach(() => { + db = new OrchestrationDb(':memory:') + }) + afterEach(() => db.close()) + + function importInstruction(target: OrchestrationDb, dispatchId: string, messageId: string): void { + expect( + importFederatedControlMessage(target, { + dispatchId, + messageId, + payload: JSON.stringify({ from: 'home', subject: 'Instruction', body: '', type: 'status' }) + }) + ).toEqual({ imported: true, type: 'status' }) + } + + function attachWithoutRunId(target: OrchestrationDb, dispatchId: string, runId?: string): void { + target.createRemoteDispatchAttachment({ + dispatchId, + runId, + taskId: `task_${dispatchId}`, + homePeerFingerprint: 'home', + protocolVersion: ORCHESTRATION_CONTRACT_VERSION, + runtimeEpoch: 'epoch', + mutationReceipt: { + callerFingerprint: 'home', + requestId: `request_${dispatchId}`, + method: 'orchestration.federationAttachStart', + payloadHash: `payload_${dispatchId}` + } + }) + } + + it('backfills a pre-upgrade attachment with a stub home Run that keeps its mailbox', () => { + db.db.exec('ALTER TABLE remote_dispatch_attachments DROP COLUMN home_run_id') + db.db.exec(`INSERT INTO remote_dispatch_attachments + (dispatch_id, task_id, home_peer_fingerprint, runtime_epoch) + VALUES ('ctx_old', 'task_old', 'home', 'epoch')`) + migrateV40.call(db, 39) + const stubRunId = federatedStubHomeRunId('ctx_old') + expect(db.getRemoteDispatchAttachment('ctx_old')?.home_run_id).toBe(stubRunId) + expect(db.getRunRaw(stubRunId)).toMatchObject({ home_database: 'remote', legacy: 0 }) + importInstruction(db, 'ctx_old', 'message_old') + expect(db.getMessageById('message_old')?.run_id).toBe(stubRunId) + }) + + it('mints a stub home Run when a v1.4.198 coordinator attaches without a Run id', () => { + attachWithoutRunId(db, 'ctx_legacy_home') + const stubRunId = federatedStubHomeRunId('ctx_legacy_home') + expect(db.getRemoteDispatchAttachment('ctx_legacy_home')?.home_run_id).toBe(stubRunId) + importInstruction(db, 'ctx_legacy_home', 'message_legacy_home') + expect(db.getMessageById('message_legacy_home')?.run_id).toBe(stubRunId) + }) + + it('rejects a whitespace-only Run id instead of minting a stub', () => { + expect(() => attachWithoutRunId(db, 'ctx_blank', ' ')).toThrow('Missing Run ID') + expect(db.getRemoteDispatchAttachment('ctx_blank')).toBeUndefined() + }) + + it('repairs rows a rolled-back v1.4.198 host inserted after user_version reached 40', () => { + const dir = mkdtempSync(join(tmpdir(), 'orca-federated-home-run-')) + const dbPath = join(dir, 'orchestration.db') + try { + const upgraded = new OrchestrationDb(dbPath) + expect(upgraded.db.pragma('user_version', { simple: true })).toBe(SCHEMA_VERSION) + // v1.4.198's insert shape: no home_run_id column, so the DEFAULT '' lands. + upgraded.db.exec(`INSERT INTO remote_dispatch_attachments + (dispatch_id, task_id, home_peer_fingerprint, runtime_epoch) + VALUES ('ctx_rolled_back', 'task_rolled_back', 'home', 'epoch')`) + expect(upgraded.getRemoteDispatchAttachment('ctx_rolled_back')?.home_run_id).toBe('') + expect(() => + importFederatedControlMessage(upgraded, { + dispatchId: 'ctx_rolled_back', + messageId: 'message_refused', + payload: JSON.stringify({ from: 'home', subject: 'x', body: '', type: 'status' }) + }) + ).toThrow(/Run not found/) + upgraded.close() + + const reopened = new OrchestrationDb(dbPath) + try { + const stubRunId = federatedStubHomeRunId('ctx_rolled_back') + expect(reopened.getRemoteDispatchAttachment('ctx_rolled_back')?.home_run_id).toBe(stubRunId) + expect(reopened.getRunRaw(stubRunId)).toMatchObject({ home_database: 'remote', legacy: 0 }) + importInstruction(reopened, 'ctx_rolled_back', 'message_rolled_back') + expect(reopened.getMessageById('message_rolled_back')?.run_id).toBe(stubRunId) + } finally { + reopened.close() + } + } finally { + rmSync(dir, { recursive: true, force: true }) + } + }) +}) diff --git a/src/main/runtime/orchestration/db/schema/migrate-v40.ts b/src/main/runtime/orchestration/db/schema/migrate-v40.ts new file mode 100644 index 00000000000..e3cd46ca30d --- /dev/null +++ b/src/main/runtime/orchestration/db/schema/migrate-v40.ts @@ -0,0 +1,15 @@ +import type { OrchestrationDb } from '../orchestration-db' +import { backfillFederatedStubHomeRuns } from '../federation/federated-stub-home-run-backfill' + +export function migrateV40(this: OrchestrationDb, current: number): void { + if (current >= 40) { + return + } + if (!this.hasColumn('remote_dispatch_attachments', 'home_run_id')) { + this.db.exec( + "ALTER TABLE remote_dispatch_attachments ADD COLUMN home_run_id TEXT NOT NULL DEFAULT ''" + ) + } + // Why: workers attached by v1.4.198 keep a mailbox; without a Run their control mail is refused. + backfillFederatedStubHomeRuns(this.db) +} diff --git a/src/main/runtime/orchestration/db/schema/migrate.ts b/src/main/runtime/orchestration/db/schema/migrate.ts index b8da910722d..582dedf4752 100644 --- a/src/main/runtime/orchestration/db/schema/migrate.ts +++ b/src/main/runtime/orchestration/db/schema/migrate.ts @@ -10,6 +10,7 @@ import { migrateV36 } from './migrate-v36' import { migrateV37 } from './migrate-v37' import { migrateV38 } from './migrate-v38' import { migrateV39 } from './migrate-v39' +import { migrateV40 } from './migrate-v40' // Why: CREATE TABLE IF NOT EXISTS won't alter existing DBs; migrate in a txn that bumps user_version only on success (atomic all-or-nothing). export function migrate(this: OrchestrationDb): void { @@ -30,6 +31,7 @@ export function migrate(this: OrchestrationDb): void { migrateV37.call(this, current) migrateV38.call(this, current) migrateV39.call(this, current) + migrateV40.call(this, current) this.createMailboxDeliveryIndexesIfPossible() this.db.pragma(`user_version = ${SCHEMA_VERSION}`) this.db.exec('COMMIT') diff --git a/src/main/runtime/orchestration/db/tasks/task-status-transition.ts b/src/main/runtime/orchestration/db/tasks/task-status-transition.ts index e1de8dc5b17..dcddc781b6d 100644 --- a/src/main/runtime/orchestration/db/tasks/task-status-transition.ts +++ b/src/main/runtime/orchestration/db/tasks/task-status-transition.ts @@ -35,18 +35,24 @@ export function updateTaskStatus( ORDER BY rowid DESC LIMIT 1` ) .get(id) as { id: string } | undefined - const activeWorker = terminalStatus - ? (this.db - .prepare( - `SELECT active.id + // Why: a supervised worker owns its Task for as long as it is alive. Every status this + // function lets past the active-Dispatch check must clear the same worker check, or the Task + // re-opens under a worker whose own lifecycle can no longer settle it (#16904 relay wedge). + // A no-op re-assert of `dispatched` re-opens nothing and stays legal. + const reopensUnderWorker = requiresActiveDispatch && task.status !== 'dispatched' + const activeWorker = + terminalStatus || reopensUnderWorker + ? (this.db + .prepare( + `SELECT active.id FROM dispatch_contexts active JOIN worker_dispatches worker ON worker.dispatch_id = active.id WHERE active.task_id = ? AND active.status IN ('pending', 'dispatched') AND worker.state NOT IN ('failed', 'succeeded', 'stopped', 'abandoned') ORDER BY active.rowid DESC LIMIT 1` - ) - .get(id) as { id: string } | undefined) - : undefined + ) + .get(id) as { id: string } | undefined) + : undefined if (activeWorker) { throw new OrchestrationError( 'task_not_startable', diff --git a/src/main/runtime/orchestration/db/tasks/task-store.ts b/src/main/runtime/orchestration/db/tasks/task-store.ts index 4ad3e8e3ffa..af43dc2be12 100644 --- a/src/main/runtime/orchestration/db/tasks/task-store.ts +++ b/src/main/runtime/orchestration/db/tasks/task-store.ts @@ -1,7 +1,6 @@ import type Database from '../../../../sqlite/sync-database' import type { TaskStatus, TaskRow } from '../../types' import { buildOrchestrationTaskDisplayMetadata } from '../../../../../shared/orchestration-task-display' -import { LEGACY_RUN_ID } from '../contract-constants' import { generateId } from '../generated-id' import type { TaskRuntimeLineageRow } from '../run-list-page' import type { OrchestrationDb } from '../orchestration-db' @@ -25,7 +24,10 @@ export function createTask( runId?: string } ): TaskRow { - const runId = task.runId ?? LEGACY_RUN_ID + const runId = task.runId + if (!runId) { + throw new Error('Run is required') + } this.requireRun(runId) if (task.parentId) { const parent = this.getTask(task.parentId) diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-authority.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-authority.ts index b89468c77a0..57ec354a8cc 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-authority.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-authority.ts @@ -15,9 +15,9 @@ export function prepareStartingWorkerAuthority( effects: unknown[] setupState: string hostScope?: string | null - // 'created': this worker-start operation created the agent terminal (including agent-first - // worktree creation, whose effects receipt says 'reused_agent_terminal'). 'external': an - // explicit --terminal reuse; ownership transfers only from an exact owned settled resource. + // 'created': this worker-start operation created the agent terminal (agent-first worktree + // creation included; its pre-rename effects rows said 'reused_agent_terminal'). 'external': + // an explicit --terminal reuse; ownership transfers only from an exact owned settled resource. terminalOwnership?: 'created' | 'external' } ): string { @@ -160,12 +160,66 @@ export function prepareStartingWorkerAuthority( } } +/** + * Custody for an agent terminal this worker-start just created, recorded at creation instead of + * after the agent boot wait. Until the row exists a keystroke into the booting pane finds no + * ownership to flip, so the takeover is silently dropped and a later `worker-release` closes the + * pane under the user. + * + * Ownership of a pane only; the Dispatch capability stays behind the boot wait, because authority + * must not be handed to a process that has not come up. + */ +export function recordCreatedWorkerTerminalCustody( + this: OrchestrationDb, + params: { + dispatchId: string + handle: string + paneKey: string + processIncarnation: string + worktreeId: string + hostScope?: string | null + } +): void { + this.db.exec('BEGIN IMMEDIATE') + try { + // Same guard as prepareStartingWorkerAuthority, read inside the transaction: a dispatch stopped + // while the terminal was being created must not acquire an owner. + const dispatch = this.getDispatchContextById(params.dispatchId) + const worker = this.getWorkerDispatch(params.dispatchId) + if (!dispatch || dispatch.status !== 'pending' || worker?.state !== 'starting') { + throw new OrchestrationError( + 'dispatch_inactive', + `Dispatch ${params.dispatchId} is not starting.` + ) + } + if (!this.getWorkerTerminalResourceByOwner(params.dispatchId)) { + this.createWorkerTerminalResourceStatement({ + dispatchId: params.dispatchId, + worktreeId: params.worktreeId, + terminalHandle: params.handle, + paneKey: params.paneKey, + processIncarnation: params.processIncarnation, + endpointId: worker.runtime_epoch, + endpointIncarnation: params.processIncarnation, + hostScope: params.hostScope, + ownership: 'owned' + }) + } + this.db.exec('COMMIT') + } catch (error) { + this.db.exec('ROLLBACK') + throw error + } +} + export type WorkerDispatchAuthorityMethods = { prepareStartingWorkerAuthority: typeof prepareStartingWorkerAuthority + recordCreatedWorkerTerminalCustody: typeof recordCreatedWorkerTerminalCustody } export function attachWorkerDispatchAuthority(ctor: { prototype: object }): void { Object.assign(ctor.prototype, { - prepareStartingWorkerAuthority + prepareStartingWorkerAuthority, + recordCreatedWorkerTerminalCustody }) } diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts index 5ff97f83fd9..99873949f92 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts @@ -2,10 +2,7 @@ import type { WorkerDispatchRow } from '../../types' import { OrchestrationError } from '../../orchestration-error' import type { OrchestrationDb } from '../orchestration-db' import { transitionLifecycleWithDb } from '../lifecycle-transition' -import { - adoptFailedStartTerminal, - type FailedStartTerminalAdoption -} from '../worker-terminal/failed-start-terminal-adoption' +import { recordFailedStartDispatchIdentity } from '../worker-terminal/failed-start-dispatch-identity' export function markWorkerDispatchReady( this: OrchestrationDb, @@ -51,11 +48,7 @@ export function failWorkerStart( // Why (#16095): revocation exists to stop a worker acting on a dispatch that never landed. A // prompt whose turn start went unobserved provably landed, so its worker keeps the authority its // own report needs. - options: { - retainCapability?: boolean - /** A start that died before authority attached still owns the terminal it created. */ - adoptResidualTerminal?: FailedStartTerminalAdoption - } = {} + options: { retainCapability?: boolean } = {} ): WorkerDispatchRow { this.db.exec('BEGIN IMMEDIATE') try { @@ -104,11 +97,7 @@ export function failWorkerStart( }) } this.closeQuestionsForDispatch(dispatchId) - adoptFailedStartTerminal( - this, - this.getWorkerDispatch(dispatchId) as WorkerDispatchRow, - options.adoptResidualTerminal - ) + recordFailedStartDispatchIdentity(this, this.getWorkerDispatch(dispatchId) as WorkerDispatchRow) this.db.exec('COMMIT') return this.getWorkerDispatch(dispatchId) as WorkerDispatchRow } catch (error) { @@ -121,7 +110,8 @@ export function markWorkerStartUnknown( this: OrchestrationDb, dispatchId: string, stage: string, - reason: string + reason: string, + effects?: unknown[] ): WorkerDispatchRow { this.db.exec('BEGIN IMMEDIATE') try { @@ -135,7 +125,12 @@ export function markWorkerStartUnknown( id: dispatchId, from: 'starting', to: 'start_unknown', - projection: { stage, last_error: reason, updated_at: new Date().toISOString() } + projection: { + stage, + last_error: reason, + updated_at: new Date().toISOString(), + ...(effects ? { effects: JSON.stringify(effects) } : {}) + } }) transitionLifecycleWithDb(this.db, { entity: 'dispatch', @@ -149,7 +144,7 @@ export function markWorkerStartUnknown( from: 'dispatched', to: 'blocked' }) - this.closeQuestionsForDispatch(dispatchId) + // Authority survives uncertainty, so its outstanding questions must remain answerable. this.db.exec('COMMIT') return this.getWorkerDispatch(dispatchId) as WorkerDispatchRow } catch (error) { diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-stop.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-stop.ts index 8dbda2030c5..fd4ee773bb4 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-stop.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-stop.ts @@ -59,7 +59,19 @@ export function beginWorkerStop( this.db.exec('COMMIT') return { disposition: 'already_settled', worker, dispatch } } - if (!['ready', 'start_unknown'].includes(worker.state)) { + // Why `stopping` under a DIFFERENT epoch is accepted: a stop whose runtime died mid-flight + // leaves the row here forever, and refusing the re-issue was the only operator escape + // (#16904). Re-running the stop earns the honest outcome — settled, or `stop_unknown`, from + // which the worker can be abandoned. It never asserts an exit the runtime did not observe. + // + // Why the epoch and not just the state: this runtime's own `stopping` row means its stop is + // still in flight, and a second pass would record `stop_unknown` over it. The exit event that + // follows claims a clean stop only from `stopping` under its own epoch + // (failActiveDispatchOnExit), so it would then read the operator's stop as a crash and + // escalate it. Same predicate as that reader, so both agree on whose stop this is. + const stopStrandedByAnotherRuntime = + worker.state === 'stopping' && worker.runtime_epoch !== runtimeEpoch + if (!['ready', 'start_unknown'].includes(worker.state) && !stopStrandedByAnotherRuntime) { throw new OrchestrationError( 'dispatch_inactive', `Dispatch ${dispatchId} cannot stop from ${worker.state}.` diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-terminal-recovery.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-terminal-recovery.ts index 97179eb0185..410318bf9c6 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-terminal-recovery.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-terminal-recovery.ts @@ -9,7 +9,6 @@ import { DISPATCH_CIRCUIT_BREAK_FAILURES } from '../dispatch-context/dispatch-ci import type { OrchestrationDb } from '../orchestration-db' import { reconcileTaskAfterDispatchInterruption } from '../dispatch-context/task-dispatch-reconciliation' import { transitionLifecycleWithDb } from '../lifecycle-transition' -import { WORKER_SETTLED_STATES } from '../../worker-terminal-ownership' export function listLegacyWorkerTerminalRecoveryRows( this: OrchestrationDb @@ -23,18 +22,9 @@ export function listLegacyWorkerTerminalRecoveryRows( FROM dispatch_contexts dc INNER JOIN worker_dispatches wd ON wd.dispatch_id = dc.id WHERE wd.state IN ('starting', 'ready', 'start_unknown', 'stopping', 'stop_unknown') - -- A settled worker whose terminal orchestration still owns keeps a resumable agent - -- session; it needs the resume fence until release or retain retires the pane. - OR (wd.state IN (${WORKER_SETTLED_STATES.map(() => '?').join(', ')}) - AND EXISTS ( - SELECT 1 FROM worker_terminal_resources wtr - WHERE wtr.owner_dispatch_id = dc.id - AND wtr.ownership_state = 'owned' - AND wtr.release_state NOT IN ('released', 'retained') - )) ORDER BY dc.rowid` ) - .all(...WORKER_SETTLED_STATES) as LegacyWorkerTerminalRecoveryRow[] + .all() as LegacyWorkerTerminalRecoveryRow[] } export function reconcileMissingWorkerTerminal( diff --git a/src/main/runtime/orchestration/db/worker-terminal/failed-start-dispatch-identity.ts b/src/main/runtime/orchestration/db/worker-terminal/failed-start-dispatch-identity.ts new file mode 100644 index 00000000000..671b12c39a7 --- /dev/null +++ b/src/main/runtime/orchestration/db/worker-terminal/failed-start-dispatch-identity.ts @@ -0,0 +1,34 @@ +import type { WorkerDispatchRow } from '../../types' +import type { OrchestrationDb } from '../orchestration-db' + +/** + * A start that dies before `prepareStartingWorkerAuthority` never filled the Dispatch context in, + * and release re-proves identity through it — so the custody row written at terminal creation would + * name a pane no release path could match. Copy that identity across. + * + * `capability_hash` stays null, so this grants nothing: it records which pane the Dispatch owns. + * + * No transaction: composes inside `failWorkerStart`'s. + */ +export function recordFailedStartDispatchIdentity( + db: OrchestrationDb, + worker: WorkerDispatchRow +): void { + const resource = db.getWorkerTerminalResourceByOwner(worker.dispatch_id) + if (!resource || resource.terminal_handle !== worker.agent_terminal_handle) { + return + } + db.db + .prepare( + `UPDATE dispatch_contexts + SET assignee_handle = ?, assignee_pane_key = ?, process_incarnation = ?, host_scope = ? + WHERE id = ? AND status = 'failed' AND capability_hash IS NULL` + ) + .run( + resource.terminal_handle, + resource.pane_key, + resource.process_incarnation, + resource.host_scope, + worker.dispatch_id + ) +} diff --git a/src/main/runtime/orchestration/db/worker-terminal/failed-start-terminal-adoption.ts b/src/main/runtime/orchestration/db/worker-terminal/failed-start-terminal-adoption.ts deleted file mode 100644 index 607ae9f45a7..00000000000 --- a/src/main/runtime/orchestration/db/worker-terminal/failed-start-terminal-adoption.ts +++ /dev/null @@ -1,68 +0,0 @@ -import type { WorkerDispatchRow } from '../../types' -import type { OrchestrationDb } from '../orchestration-db' - -/** Identity of a terminal this worker-start created and never handed to an owner. */ -export type FailedStartTerminalAdoption = { - terminalHandle: string - worktreeId: string | null - paneKey: string - processIncarnation: string - hostScope?: string | null -} - -/** - * A start that dies before `prepareStartingWorkerAuthority` leaves the terminal it created with no - * owner, so no release path can ever close it and the fleet can only say `inspect`. Record the - * ownership the successful path would have recorded, so ordinary `worker-release` owns the cleanup. - * - * No transaction: composes inside `failWorkerStart`'s. - */ -export function adoptFailedStartTerminal( - db: OrchestrationDb, - worker: WorkerDispatchRow, - adoption: FailedStartTerminalAdoption | undefined -): void { - if (!adoption || worker.agent_terminal_handle !== adoption.terminalHandle) { - return - } - if (db.getWorkerTerminalResourceByOwner(worker.dispatch_id)) { - return - } - // A second owner for one process could close it twice, or close a terminal already handed on. - const conflict = db.db - .prepare( - `SELECT 1 FROM worker_terminal_resources - WHERE ownership_state <> 'released' - AND (terminal_handle = ? OR process_incarnation = ?) LIMIT 1` - ) - .get(adoption.terminalHandle, adoption.processIncarnation) - if (conflict) { - return - } - db.createWorkerTerminalResourceStatement({ - dispatchId: worker.dispatch_id, - worktreeId: adoption.worktreeId ?? worker.worktree_id, - terminalHandle: adoption.terminalHandle, - paneKey: adoption.paneKey, - processIncarnation: adoption.processIncarnation, - endpointId: worker.runtime_epoch ?? null, - endpointIncarnation: adoption.processIncarnation, - hostScope: adoption.hostScope ?? null, - ownership: 'owned' - }) - // Release re-proves identity through the Dispatch context, which a failed start never filled in. - // This records which pane the Dispatch owns; `capability_hash` stays null, so it grants nothing. - db.db - .prepare( - `UPDATE dispatch_contexts - SET assignee_handle = ?, assignee_pane_key = ?, process_incarnation = ?, host_scope = ? - WHERE id = ? AND status = 'failed' AND capability_hash IS NULL` - ) - .run( - adoption.terminalHandle, - adoption.paneKey, - adoption.processIncarnation, - adoption.hostScope ?? null, - worker.dispatch_id - ) -} diff --git a/src/main/runtime/orchestration/db/worker-terminal/worker-terminal-resource-store.ts b/src/main/runtime/orchestration/db/worker-terminal/worker-terminal-resource-store.ts index 1d7232107b1..e6942503dbf 100644 --- a/src/main/runtime/orchestration/db/worker-terminal/worker-terminal-resource-store.ts +++ b/src/main/runtime/orchestration/db/worker-terminal/worker-terminal-resource-store.ts @@ -2,6 +2,7 @@ import type { WorkerTerminalResourceRow, WorkerTerminalOwnershipState } from '../../worker-terminal-ownership' +import { WORKER_SETTLED_STATES } from '../../worker-terminal-ownership' import { OrchestrationError } from '../../orchestration-error' import { generateId } from '../generated-id' import type { OrchestrationDb } from '../orchestration-db' @@ -199,9 +200,40 @@ export function transferWorkerTerminalResourceStatement( return this.getWorkerTerminalResource(params.resourceId) as WorkerTerminalResourceRow } +// A new process in the same pane is ordinary user work, not the settled Dispatch's resource. +export function retainReplacedWorkerTerminalResources( + this: OrchestrationDb, + params: { paneKey: string; worktreeId: string; hostScope: string; processIncarnation: string } +): number { + return Number( + this.db + .prepare( + `UPDATE worker_terminal_resources + SET release_state = 'retained', retained_reason = 'identity_unproven', + updated_at = datetime('now') + WHERE pane_key = ? AND worktree_id = ? AND host_scope = ? + AND process_incarnation IS NOT NULL AND process_incarnation != ? + AND ownership_state = 'owned' AND release_state = 'not_requested' + AND EXISTS ( + SELECT 1 FROM worker_dispatches w + WHERE w.dispatch_id = worker_terminal_resources.owner_dispatch_id + AND w.state IN (${WORKER_SETTLED_STATES.map(() => '?').join(', ')}) + )` + ) + .run( + params.paneKey, + params.worktreeId, + params.hostScope, + params.processIncarnation, + ...WORKER_SETTLED_STATES + ).changes + ) +} + // Finds an owned, settled, exact-match resource for an explicitly reused terminal. export type WorkerTerminalResourceStoreMethods = { + retainReplacedWorkerTerminalResources: typeof retainReplacedWorkerTerminalResources backfillWorkerTerminalResources: typeof backfillWorkerTerminalResources createWorkerTerminalResourceStatement: typeof createWorkerTerminalResourceStatement getWorkerTerminalResource: typeof getWorkerTerminalResource @@ -214,6 +246,7 @@ export type WorkerTerminalResourceStoreMethods = { export function attachWorkerTerminalResourceStore(ctor: { prototype: object }): void { Object.assign(ctor.prototype, { + retainReplacedWorkerTerminalResources, backfillWorkerTerminalResources, createWorkerTerminalResourceStatement, getWorkerTerminalResource, diff --git a/src/main/runtime/orchestration/db/writer-run-required.test.ts b/src/main/runtime/orchestration/db/writer-run-required.test.ts new file mode 100644 index 00000000000..682fabb1b60 --- /dev/null +++ b/src/main/runtime/orchestration/db/writer-run-required.test.ts @@ -0,0 +1,41 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { OrchestrationDb } from './orchestration-db' +import { UNBOUND_RUN_ID } from './contract-constants' + +describe('writers require a Run', () => { + let db: OrchestrationDb + beforeEach(() => { + db = new OrchestrationDb(':memory:') + }) + afterEach(() => db.close()) + + it('files a message without a Run under the unbound Run, never the legacy one', () => { + const message = db.insertMessage({ from: 'sender', to: 'worker', subject: 'mail' }) + expect(message.run_id).toBe(UNBOUND_RUN_ID) + expect(db.getRun(UNBOUND_RUN_ID)).toMatchObject({ legacy: 0 }) + expect(db.getUnreadMessages('worker').map((row) => row.id)).toEqual([message.id]) + }) + + it('rejects a Task without a Run instead of using the legacy Run', () => { + expect(() => db.createTask({ spec: 'work' })).toThrow('Run is required') + expect(db.listTasks()).toEqual([]) + }) + + it('rejects a decision gate whose Task has no Run', () => { + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) + vi.spyOn(db, 'getTask').mockReturnValue({ ...task, run_id: undefined } as never) + expect(() => db.createGate({ taskId: task.id, question: 'Proceed?' })).toThrow() + expect(db.listGates()).toEqual([]) + }) + + it('rejects a decision gate without a Task before writing', () => { + db.db.exec(` + CREATE TRIGGER reject_gate_insert BEFORE INSERT ON decision_gates + BEGIN SELECT RAISE(ABORT, 'gate insert reached'); END; + `) + expect(() => db.createGate({ taskId: 'missing', question: 'Proceed?' })).toThrow( + 'Task missing was not found while creating a decision gate.' + ) + expect(db.listGates()).toEqual([]) + }) +}) diff --git a/src/main/runtime/orchestration/dispatch-failure-idempotency.test.ts b/src/main/runtime/orchestration/dispatch-failure-idempotency.test.ts index c6f75723cf3..5704db1490e 100644 --- a/src/main/runtime/orchestration/dispatch-failure-idempotency.test.ts +++ b/src/main/runtime/orchestration/dispatch-failure-idempotency.test.ts @@ -6,7 +6,7 @@ import { createRootDispatch } from './db/root-dispatch-test-fixture' describe('dispatch failure idempotency', () => { it('counts an active dispatch failure only once', () => { const db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') expect(db.failDispatch(dispatch.id, 'exit')?.failure_count).toBe(1) @@ -19,7 +19,7 @@ describe('dispatch failure idempotency', () => { it('does not overwrite a completed dispatch', () => { const db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') db.completeDispatch(dispatch.id) @@ -33,7 +33,7 @@ describe('dispatch failure idempotency', () => { it('rolls back the dispatch when the task update fails', () => { const db = new OrchestrationDb(':memory:') const sqlite = (db as unknown as { db: Database.Database }).db - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') sqlite.exec(` CREATE TRIGGER reject_task_failure_update diff --git a/src/main/runtime/orchestration/failed-start-terminal-adoption.test.ts b/src/main/runtime/orchestration/failed-start-terminal-adoption.test.ts deleted file mode 100644 index d38248f9cb8..00000000000 --- a/src/main/runtime/orchestration/failed-start-terminal-adoption.test.ts +++ /dev/null @@ -1,157 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import { OrchestrationDb } from './db' - -const HANDLE = 'term_residual' -const PANE_KEY = 'tab_residual:leaf_residual' -const INCARNATION = 'runtime:pty-residual:1' - -describe('a start that fails before authority still owns the terminal it created', () => { - let db: OrchestrationDb | undefined - - afterEach(() => { - db?.close() - }) - - /** Replays the shipping order: readiness stage records the handle, then the wait fails. */ - function failStartAfterCreatingTerminal( - adoption?: Parameters[3] - ): { db: OrchestrationDb; dispatchId: string } { - const d = (db = new OrchestrationDb(':memory:')) - const task = d.createTask({ spec: 'residual terminal' }) - const started = d.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: task.id, - startOptions: {} - }) - const effects = [ - { kind: 'terminal', role: 'agent', action: 'created', id: HANDLE, surface: 'visible' } - ] - d.recordWorkerStage({ - dispatchId: started.dispatch.id, - stage: 'terminal_readying', - worktreeId: 'repo::worktree', - terminalHandle: HANDLE, - effects, - residualResources: effects - }) - d.failWorkerStart( - started.dispatch.id, - 'agent_readiness', - 'Agent startup blocked: codex-interactive-prompt', - adoption - ) - return { db: d, dispatchId: started.dispatch.id } - } - - const adoption = { - adoptResidualTerminal: { - terminalHandle: HANDLE, - worktreeId: 'repo::worktree', - paneKey: PANE_KEY, - processIncarnation: INCARNATION, - hostScope: null - } - } - - it('leaves nothing that can close the terminal when the start is not adopted', () => { - const { db: d, dispatchId } = failStartAfterCreatingTerminal() - - expect(d.getWorkerTerminalResourceByOwner(dispatchId)).toBeUndefined() - expect(d.requestWorkerTerminalRelease(dispatchId)).toMatchObject({ - disposition: 'retained', - reason: 'no_owned_resource' - }) - }) - - it('records the ownership the successful path would have recorded', () => { - const { db: d, dispatchId } = failStartAfterCreatingTerminal(adoption) - - expect(d.getWorkerTerminalResourceByOwner(dispatchId)).toMatchObject({ - owner_dispatch_id: dispatchId, - terminal_handle: HANDLE, - pane_key: PANE_KEY, - process_incarnation: INCARNATION, - ownership_state: 'owned', - release_state: 'not_requested' - }) - }) - - it('lets worker-release proceed on the failed dispatch', () => { - const { db: d, dispatchId } = failStartAfterCreatingTerminal(adoption) - - expect(d.requestWorkerTerminalRelease(dispatchId)).toMatchObject({ - disposition: 'requested', - resource: { release_state: 'requested' } - }) - }) - - it('re-proves identity through the dispatch context release reads', () => { - const { db: d, dispatchId } = failStartAfterCreatingTerminal(adoption) - - expect( - d.isDispatchProcessCurrent({ dispatchId, paneKey: PANE_KEY, processIncarnation: INCARNATION }) - ).toBe(true) - // Adoption records which pane the dispatch owns; it never restores authority over it. - expect(d.getDispatchContextById(dispatchId)).toMatchObject({ - status: 'failed', - capability_hash: null - }) - expect(d.getDispatchContextById(dispatchId)?.capability_revoked_at).not.toBeNull() - }) - - it('publishes the terminal as reclaimable so the fleet names release', () => { - const { db: d, dispatchId } = failStartAfterCreatingTerminal(adoption) - - expect(d.listWorkerTerminalResources({ dispatchIds: [dispatchId] })[0]).toMatchObject({ - agentTerminalHandle: HANDLE, - terminalState: 'reclaimable' - }) - }) - - it('never claims a terminal the durable row does not name', () => { - const { db: d, dispatchId } = failStartAfterCreatingTerminal({ - adoptResidualTerminal: { ...adoption.adoptResidualTerminal, terminalHandle: 'term_other' } - }) - - expect(d.getWorkerTerminalResourceByOwner(dispatchId)).toBeUndefined() - }) - - it('never claims a terminal another live resource already accounts for', () => { - const d = (db = new OrchestrationDb(':memory:')) - const first = d.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: d.createTask({ spec: 'owner' }).id, - startOptions: {} - }) - d.prepareStartingWorkerAuthority({ - dispatchId: first.dispatch.id, - handle: HANDLE, - paneKey: PANE_KEY, - processIncarnation: INCARNATION, - worktreeId: 'repo::worktree', - setupState: 'not_applicable', - effects: [], - terminalOwnership: 'created' - }) - const second = d.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: d.createTask({ spec: 'claimant' }).id, - startOptions: {} - }) - d.recordWorkerStage({ - dispatchId: second.dispatch.id, - stage: 'terminal_readying', - terminalHandle: HANDLE - }) - - d.failWorkerStart(second.dispatch.id, 'agent_readiness', 'blocked', adoption) - - expect(d.getWorkerTerminalResourceByOwner(second.dispatch.id)).toBeUndefined() - expect(d.getWorkerTerminalResourceByOwner(first.dispatch.id)).toMatchObject({ - ownership_state: 'owned' - }) - }) -}) diff --git a/src/main/runtime/orchestration/federation-acknowledgment-integrity.test.ts b/src/main/runtime/orchestration/federation-acknowledgment-integrity.test.ts index fd8e32d1a32..e5c452f91a1 100644 --- a/src/main/runtime/orchestration/federation-acknowledgment-integrity.test.ts +++ b/src/main/runtime/orchestration/federation-acknowledgment-integrity.test.ts @@ -14,6 +14,7 @@ describe('federation acknowledgment integrity', () => { db = new OrchestrationDb(':memory:') const dispatchId = `ctx_protocol_${protocolVersion}` db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: `task_protocol_${protocolVersion}`, homePeerFingerprint: 'home_peer', diff --git a/src/main/runtime/orchestration/federation-control-message.ts b/src/main/runtime/orchestration/federation-control-message.ts index bcab04f99b8..2d86ef3c67b 100644 --- a/src/main/runtime/orchestration/federation-control-message.ts +++ b/src/main/runtime/orchestration/federation-control-message.ts @@ -58,11 +58,20 @@ export function importFederatedControlMessage( payload: string } ): { imported: boolean; type: MessageType } { + const attachment = db.getRemoteDispatchAttachment(params.dispatchId) + if (!attachment) { + throw new OrchestrationError( + 'dispatch_not_found', + `Remote Dispatch ${params.dispatchId} was not found.` + ) + } + db.requireRun(attachment.home_run_id) const message = parseFederatedControlMessage(params.payload) const recipient = `dispatch:${params.dispatchId}` const existing = db.getMessageById(params.messageId) if (existing) { if ( + existing.run_id !== attachment.home_run_id || existing.to_handle !== recipient || existing.from_handle !== message.from || existing.subject !== message.subject || @@ -81,6 +90,7 @@ export function importFederatedControlMessage( } db.insertMessage({ id: params.messageId, + runId: attachment.home_run_id, from: message.from, to: recipient, subject: message.subject, diff --git a/src/main/runtime/orchestration/lifecycle-caller-edges.test.ts b/src/main/runtime/orchestration/lifecycle-caller-edges.test.ts index 3bc2eee4ae9..7f445388a25 100644 --- a/src/main/runtime/orchestration/lifecycle-caller-edges.test.ts +++ b/src/main/runtime/orchestration/lifecycle-caller-edges.test.ts @@ -109,7 +109,10 @@ describe('lifecycle graph against its callers', () => { it('settles a stopping worker whose PTY exits during the stop', () => { const database = createDatabase() - const task = database.createTask({ spec: 'stopping exited worker' }) + const task = database.createTask({ + runId: 'run_legacy_local', + spec: 'stopping exited worker' + }) const dispatchId = startWorker(database, task.id, 'stopping_exited') expect(database.beginWorkerStop(dispatchId, 'runtime_test').disposition).toBe('stopping') @@ -127,9 +130,18 @@ describe('lifecycle graph against its callers', () => { it('still lets a coordinator reopen or overturn a settled Task', () => { const database = createDatabase() - const reopened = database.createTask({ spec: 'reopen me' }) - const overturned = database.createTask({ spec: 'overturn me' }) - const retried = database.createTask({ spec: 'retry me' }) + const reopened = database.createTask({ + runId: 'run_legacy_local', + spec: 'reopen me' + }) + const overturned = database.createTask({ + runId: 'run_legacy_local', + spec: 'overturn me' + }) + const retried = database.createTask({ + runId: 'run_legacy_local', + spec: 'retry me' + }) database.updateTaskStatus(reopened.id, 'completed', 'first result') database.updateTaskStatus(overturned.id, 'completed', 'wrong result') database.updateTaskStatus(retried.id, 'failed', 'boom') diff --git a/src/main/runtime/orchestration/lifecycle-reconciliation.test.ts b/src/main/runtime/orchestration/lifecycle-reconciliation.test.ts index 3f0f0a7664f..dedea449629 100644 --- a/src/main/runtime/orchestration/lifecycle-reconciliation.test.ts +++ b/src/main/runtime/orchestration/lifecycle-reconciliation.test.ts @@ -10,10 +10,11 @@ describe('lifecycle reconciliation', () => { it('rejects handle churn when neither side has stable pane identity', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_before_restart') const logs: string[] = [] const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_after_restart', to: 'term_coordinator', subject: 'Done', @@ -37,9 +38,10 @@ describe('lifecycle reconciliation', () => { it('completes worker_done from the dispatched pane after a handle remint', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_before_restart', `tab_w:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_after_restart', to: 'term_coordinator', subject: 'Done', @@ -54,7 +56,7 @@ describe('lifecycle reconciliation', () => { it('completes an exact-authority worker_done after an uncertain worker start', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -82,6 +84,7 @@ describe('lifecycle reconciliation', () => { ).toEqual({ valid: true }) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Done after reconnect', @@ -106,9 +109,10 @@ describe('lifecycle reconciliation', () => { it('fails both the dispatch and task from an authenticated failed worker report', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker', `tab_w:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Failed: tests cannot start', @@ -139,7 +143,7 @@ describe('lifecycle reconciliation', () => { it('keeps worker report settlement nested in its caller transaction', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') db.db.exec('BEGIN IMMEDIATE') @@ -160,19 +164,16 @@ describe('lifecycle reconciliation', () => { it('replays an identical terminal outcome without mutating settled state', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') const makeMessage = () => db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Done', type: 'worker_done', - payload: JSON.stringify({ - taskId: task.id, - dispatchId: dispatch.id, - outcome: 'succeeded' - }) + payload: JSON.stringify({ taskId: task.id, dispatchId: dispatch.id, outcome: 'succeeded' }) }) expect(reconcileLifecycleMessage(db, makeMessage()).action).toBe('completed') @@ -199,6 +200,7 @@ describe('lifecycle reconciliation', () => { ])('rejects malformed worker reports with $code', ({ payload, code }) => { db = new OrchestrationDb(':memory:') const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Done', @@ -215,11 +217,12 @@ describe('lifecycle reconciliation', () => { it('completes worker_done from the same leaf after a pane break-out changed the tab half', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) // Dispatch recorded the post-break-out pane key; the worker shell still // holds the spawn-time key with the old tab id. const dispatch = createRootDispatch(db, task.id, 'term_before_restart', `tab_new:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_after_restart', to: 'term_coordinator', subject: 'Done', @@ -234,9 +237,10 @@ describe('lifecycle reconciliation', () => { it('rejects mismatched opaque pane keys instead of treating them as legacy', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', `tab_w:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_reminted', to: 'term_coordinator', subject: 'Done', @@ -251,9 +255,10 @@ describe('lifecycle reconciliation', () => { it('rejects worker_done from a foreign pane that claims the assignee handle', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', `tab_w1:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_owner', to: 'term_coordinator', subject: 'Done', @@ -294,9 +299,10 @@ describe('lifecycle reconciliation', () => { it('does not let a caller-supplied rejection marker turn completion into success', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker', `tab_w:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Done', @@ -323,9 +329,10 @@ describe('lifecycle reconciliation', () => { it('rejects a coordinator completion for a pane-bound dispatch', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker', `tab_w:${LEAF_A}`) const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_coordinator', to: 'term_coordinator', subject: 'Done', @@ -342,9 +349,13 @@ describe('lifecycle reconciliation', () => { it('uses exact handle equality only for a legacy dispatch without a pane key', () => { db = new OrchestrationDb(':memory:') - const acceptedTask = db.createTask({ spec: 'legacy work' }) + const acceptedTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'legacy work' + }) const acceptedDispatch = createRootDispatch(db, acceptedTask.id, 'term_legacy') const accepted = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy', to: 'term_coordinator', subject: 'Done', @@ -357,9 +368,13 @@ describe('lifecycle reconciliation', () => { }) expect(reconcileLifecycleMessage(db, accepted).action).toBe('completed') - const rejectedTask = db.createTask({ spec: 'other legacy work' }) + const rejectedTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'other legacy work' + }) const rejectedDispatch = createRootDispatch(db, rejectedTask.id, 'term_other_legacy') const rejected = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_foreign', to: 'term_coordinator', subject: 'Done', @@ -379,8 +394,12 @@ describe('lifecycle reconciliation', () => { it('does not release a dependent when a foreign completion wins the arrival race', () => { db = new OrchestrationDb(':memory:') - const parent = db.createTask({ spec: 'parent' }) - const child = db.createTask({ spec: 'child', deps: [parent.id] }) + const parent = db.createTask({ runId: 'run_legacy_local', spec: 'parent' }) + const child = db.createTask({ + runId: 'run_legacy_local', + spec: 'child', + deps: [parent.id] + }) const dispatch = createRootDispatch(db, parent.id, 'term_worker', `tab_w:${LEAF_A}`) const payload = JSON.stringify({ taskId: parent.id, @@ -389,6 +408,7 @@ describe('lifecycle reconciliation', () => { }) const foreign = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_coordinator', to: 'term_coordinator', subject: 'Done', @@ -403,6 +423,7 @@ describe('lifecycle reconciliation', () => { expect(db.getTask(child.id)?.status).toBe('pending') const owner = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker_reminted', to: 'term_coordinator', subject: 'Done', @@ -416,7 +437,7 @@ describe('lifecycle reconciliation', () => { it('does not let a foreign replay overwrite an authorized completion', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker', `tab_w:${LEAF_A}`) const payload = JSON.stringify({ taskId: task.id, @@ -424,6 +445,7 @@ describe('lifecycle reconciliation', () => { outcome: 'succeeded' }) const owner = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Done', @@ -435,6 +457,7 @@ describe('lifecycle reconciliation', () => { const result = db.getTask(task.id)?.result const replay = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_foreign', to: 'term_coordinator', subject: 'Forged replay', @@ -451,10 +474,11 @@ describe('lifecycle reconciliation', () => { it('surfaces worker_done sent from a different pane as rejected', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', `tab_w1:${LEAF_A}`) const logs: string[] = [] const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_other_worker', to: 'term_coordinator', subject: 'Done', @@ -474,9 +498,10 @@ describe('lifecycle reconciliation', () => { it('surfaces a heartbeat sent from a different pane without recording liveness', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', `tab_w1:${LEAF_A}`) const heartbeat = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_other_worker', to: 'term_coordinator', subject: 'alive', @@ -508,9 +533,10 @@ describe('lifecycle reconciliation', () => { it('surfaces a foreign heartbeat that claims the assignee handle', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', `tab_w1:${LEAF_A}`) const heartbeat = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_owner', to: 'term_coordinator', subject: 'alive', @@ -528,9 +554,10 @@ describe('lifecycle reconciliation', () => { it('records a heartbeat whose pane key drifted only in the tab half', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_owner', `tab_new:${LEAF_A}`) const heartbeat = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_owner', to: 'term_coordinator', subject: 'alive', @@ -548,12 +575,16 @@ describe('lifecycle reconciliation', () => { it('suppresses same-dispatch heartbeats once worker_done is reconciled', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'work' }) const dispatch = createRootDispatch(db, task.id, 'term_worker') - const otherTask = db.createTask({ spec: 'other work' }) + const otherTask = db.createTask({ + runId: 'run_legacy_local', + spec: 'other work' + }) const otherDispatch = createRootDispatch(db, otherTask.id, 'term_other') const insertHeartbeat = (dispatchId: string, from: string) => db.insertMessage({ + runId: 'run_legacy_local', from, to: 'term_coordinator', subject: 'alive', @@ -565,6 +596,7 @@ describe('lifecycle reconciliation', () => { reconcileLifecycleMessage(db, staleHeartbeat) reconcileLifecycleMessage(db, otherHeartbeat) const done = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coordinator', subject: 'Done', diff --git a/src/main/runtime/orchestration/lightweight-run-worker-exit-escalation.test.ts b/src/main/runtime/orchestration/lightweight-run-worker-exit-escalation.test.ts index 2b8d9c90bc2..d37e379e1c5 100644 --- a/src/main/runtime/orchestration/lightweight-run-worker-exit-escalation.test.ts +++ b/src/main/runtime/orchestration/lightweight-run-worker-exit-escalation.test.ts @@ -412,7 +412,7 @@ describe('STA-4604 worker PTY exit escalation reaches the coordinator', () => { } }) - it('falls back to the legacy gate when the dispatch owning Run row is gone', async () => { + it('preserves the dispatch Run when legacy coordinator routing is used', async () => { const { runtime, workerHandle, coordinatorHandle } = makeRuntimeWithTwoPanes() const insertMessage = vi.fn((message: { to: string }) => ({ ...message, @@ -435,8 +435,7 @@ describe('STA-4604 worker PTY exit escalation reaches the coordinator', () => { expect(insertMessage).toHaveBeenCalledWith( expect.objectContaining({ to: coordinatorHandle, type: 'escalation' }) ) - // An orphaned dispatch has no Run mailbox to address, so it must not invent one. - expect(insertMessage.mock.calls[0]?.[0]).not.toHaveProperty('runId') + expect(insertMessage.mock.calls[0]?.[0]).toHaveProperty('runId', 'run-that-no-longer-exists') }) it('still reaches the Run mailbox when the Run has no bound coordinator', async () => { diff --git a/src/main/runtime/orchestration/mailbox-pointer-eligibility.test.ts b/src/main/runtime/orchestration/mailbox-pointer-eligibility.test.ts index 87b090e4e48..2276665fc1c 100644 --- a/src/main/runtime/orchestration/mailbox-pointer-eligibility.test.ts +++ b/src/main/runtime/orchestration/mailbox-pointer-eligibility.test.ts @@ -15,9 +15,9 @@ const MAILBOX = 'dispatch:d1' function seeded(): OrchestrationDb { const db = new OrchestrationDb(':memory:') db.insertMessages([ - { from: 'coordinator', to: MAILBOX, subject: 'a', type: 'status' }, - { from: 'coordinator', to: MAILBOX, subject: 'b', type: 'question' }, - { from: 'coordinator', to: MAILBOX, subject: 'c', type: 'status' } + { runId: 'run_legacy_local', from: 'coordinator', to: MAILBOX, subject: 'a', type: 'status' }, + { runId: 'run_legacy_local', from: 'coordinator', to: MAILBOX, subject: 'b', type: 'question' }, + { runId: 'run_legacy_local', from: 'coordinator', to: MAILBOX, subject: 'c', type: 'status' } ]) return db } diff --git a/src/main/runtime/orchestration/mailbox-pointer-stage.test.ts b/src/main/runtime/orchestration/mailbox-pointer-stage.test.ts index 9573f02fc0c..d76a480ecfd 100644 --- a/src/main/runtime/orchestration/mailbox-pointer-stage.test.ts +++ b/src/main/runtime/orchestration/mailbox-pointer-stage.test.ts @@ -59,7 +59,12 @@ function stageArgs(db: OrchestrationDb, state: OrchestrationMailboxPointerState) describe('mailbox pointer staging watermark', () => { it('leaves no watermark when the reservation claim is lost', () => { const db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 's' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 's' + }) // A concurrent flight already owns the reservation, so this claim cannot succeed. expect( db.stageMailboxPointerEnter([message.id], { ptyId: 'other-pty', processIncarnation: 'inc-x' }) @@ -79,7 +84,12 @@ describe('mailbox pointer staging watermark', () => { it('leaves no watermark when the reservation write throws', () => { const db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 's' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 's' + }) const throwing = new Proxy(db, { get(target, prop, receiver) { if (prop === 'markMailboxPointerWriteAttempted') { @@ -107,7 +117,12 @@ describe('mailbox pointer staging watermark', () => { it('keeps the watermark for the flight that owns the reservation', () => { const db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 's' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 's' + }) const state = new OrchestrationMailboxPointerState() const args = stageArgs(db, state) stageOrchestrationMailboxPointer({ @@ -122,7 +137,12 @@ describe('mailbox pointer staging watermark', () => { it('drains a delivery parked behind the watermark when the write is refused', () => { const db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 's' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 's' + }) const state = new OrchestrationMailboxPointerState() const args = stageArgs(db, state) const redrive = vi.fn() @@ -148,7 +168,7 @@ describe('mailbox pointer staging watermark', () => { it('still points new mail after a delivery lost its reservation claim', async () => { const db = new OrchestrationDb(':memory:') - db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'first' }) + db.insertMessage({ runId: 'run_legacy_local', from: 'a', to: 'run:run-1', subject: 'first' }) let stealNextClaim = true const contended = new Proxy(db, { get(target, prop, receiver) { @@ -172,7 +192,7 @@ describe('mailbox pointer staging watermark', () => { expect(writePty).not.toHaveBeenCalled() // Newer mail must still reach the agent; a leaked watermark used to park it forever. - db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'second' }) + db.insertMessage({ runId: 'run_legacy_local', from: 'a', to: 'run:run-1', subject: 'second' }) delivery.deliver(LEAF, { mailboxHandle: 'run:run-1', skipAbsenceProbe: true }) await new Promise((resolve) => setImmediate(resolve)) diff --git a/src/main/runtime/orchestration/mailbox-pointer-submit.test.ts b/src/main/runtime/orchestration/mailbox-pointer-submit.test.ts index 00126bc237b..02d556204df 100644 --- a/src/main/runtime/orchestration/mailbox-pointer-submit.test.ts +++ b/src/main/runtime/orchestration/mailbox-pointer-submit.test.ts @@ -14,7 +14,12 @@ import type { WriteSettlement } from '../../../shared/pty-write-settlement' describe('orchestration mailbox pointer submit', () => { it('does not settle a replacement reservation after an old Enter write resolves', async () => { const db = new OrchestrationDb(':memory:') - const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'staged' }) + const message = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'staged' + }) const ptyId = 'pty-reused' const oldReservation = { ptyId, processIncarnation: 'inc-old' } const replacementReservation = { ptyId, processIncarnation: 'inc-new' } @@ -86,8 +91,18 @@ describe('orchestration mailbox pointer submit', () => { it('does not overwrite a message already reserved by another pointer flight', () => { const db = new OrchestrationDb(':memory:') - const first = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'first' }) - const second = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'second' }) + const first = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'first' + }) + const second = db.insertMessage({ + runId: 'run_legacy_local', + from: 'a', + to: 'run:run-1', + subject: 'second' + }) const original = { ptyId: 'pty-a', processIncarnation: 'inc-a' } const replacement = { ptyId: 'pty-b', processIncarnation: 'inc-b' } diff --git a/src/main/runtime/orchestration/message-batch-atomicity.test.ts b/src/main/runtime/orchestration/message-batch-atomicity.test.ts index f2e43ed31e4..fda83fad9a9 100644 --- a/src/main/runtime/orchestration/message-batch-atomicity.test.ts +++ b/src/main/runtime/orchestration/message-batch-atomicity.test.ts @@ -108,8 +108,20 @@ describe('message batch atomicity', () => { expect(() => db?.insertMessages([ - { id: 'inner_first', from: 'sender', to: 'recipient', subject: 'first' }, - { id: 'inner_second', from: 'sender', to: 'recipient', subject: 'second' } + { + runId: 'run_legacy_local', + id: 'inner_first', + from: 'sender', + to: 'recipient', + subject: 'first' + }, + { + runId: 'run_legacy_local', + id: 'inner_second', + from: 'sender', + to: 'recipient', + subject: 'second' + } ]) ).toThrow('blocked') sqlite.exec('COMMIT') @@ -133,6 +145,7 @@ describe('message batch atomicity', () => { expect(() => db?.commitWorkerDoneMessageMutation(() => { db?.insertMessage({ + runId: 'run_legacy_local', id: 'inner', from: 'worker', to: 'coordinator', diff --git a/src/main/runtime/orchestration/nested-worker-depth-migration.test.ts b/src/main/runtime/orchestration/nested-worker-depth-migration.test.ts index fa955b7cf08..9d05c0dac07 100644 --- a/src/main/runtime/orchestration/nested-worker-depth-migration.test.ts +++ b/src/main/runtime/orchestration/nested-worker-depth-migration.test.ts @@ -34,6 +34,7 @@ describe('nested worker depth migration (v30)', () => { const oldDb = new Database(dbPath) oldDb.exec('ALTER TABLE dispatch_contexts DROP COLUMN depth') oldDb.exec('ALTER TABLE remote_dispatch_attachments DROP COLUMN depth') + oldDb.exec('ALTER TABLE remote_dispatch_attachments DROP COLUMN home_run_id') oldDb.pragma('user_version = 29') oldDb .prepare( @@ -78,7 +79,7 @@ describe('nested worker depth migration (v30)', () => { ) .run() - const task = db.createTask({ spec: 'post-upgrade nesting attempt' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'post-upgrade nesting attempt' }) expect(() => db!.createDispatchContext({ taskId: task.id, diff --git a/src/main/runtime/orchestration/orchestration-adopted-run-binding.test.ts b/src/main/runtime/orchestration/orchestration-adopted-run-binding.test.ts index 800a7511451..d667267e0a6 100644 --- a/src/main/runtime/orchestration/orchestration-adopted-run-binding.test.ts +++ b/src/main/runtime/orchestration/orchestration-adopted-run-binding.test.ts @@ -47,11 +47,13 @@ function createAdoptedFixture(options: { settleWork: boolean }): AdoptedFixture const before = new OrchestrationDb(dbPath) const task = before.createTask({ + runId: 'run_legacy_local', spec: 'legacy assignment', createdByTerminalHandle: LEGACY_COORDINATOR_HANDLE }) const dispatch = createRootDispatch(before, task.id, LEGACY_WORKER_HANDLE, LEGACY_WORKER_PANE) const recovery = before.insertMessage({ + runId: 'run_legacy_local', from: LEGACY_WORKER_HANDLE, to: LEGACY_COORDINATOR_HANDLE, subject: 'recovered worker outcome', diff --git a/src/main/runtime/orchestration/orchestration-all-start-versions-migration.test.ts b/src/main/runtime/orchestration/orchestration-all-start-versions-migration.test.ts index b4c9281d89b..b62677e465b 100644 --- a/src/main/runtime/orchestration/orchestration-all-start-versions-migration.test.ts +++ b/src/main/runtime/orchestration/orchestration-all-start-versions-migration.test.ts @@ -36,7 +36,12 @@ describe('orchestration migration from every prior version stamp', () => { expect(reopened.db.pragma('user_version', { simple: true }), `reopen v${version}`).toBe( SCHEMA_VERSION ) - expect(() => reopened.createTask({ spec: `migration v${version}` })).not.toThrow() + expect(() => + reopened.createTask({ + runId: 'run_legacy_local', + spec: `migration v${version}` + }) + ).not.toThrow() reopened.close() } }) diff --git a/src/main/runtime/orchestration/orchestration-db-retention-pagination.test.ts b/src/main/runtime/orchestration/orchestration-db-retention-pagination.test.ts index 38eeca64337..c980b8fe02a 100644 --- a/src/main/runtime/orchestration/orchestration-db-retention-pagination.test.ts +++ b/src/main/runtime/orchestration/orchestration-db-retention-pagination.test.ts @@ -103,6 +103,7 @@ describe('OrchestrationDb bounded mutation receipts', () => { insertMutationReceipts(db, MUTATION_RECEIPT_MAX_ROWS, 'completed') db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId: 'ctx_remote_pruned', taskId: 'task_remote_pruned', homePeerFingerprint: 'caller', @@ -131,6 +132,7 @@ describe('OrchestrationDb bounded mutation receipts', () => { expect(() => db!.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId: 'ctx_remote_overflow', taskId: 'task_remote_overflow', homePeerFingerprint: 'caller', @@ -151,7 +153,7 @@ describe('OrchestrationDb bounded mutation receipts', () => { it('guards atomic worker acceptance without changing task state', () => { db = new OrchestrationDb(':memory:') - const task = db.createTask({ spec: 'capacity check' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'capacity check' }) insertMutationReceipts(db, MUTATION_RECEIPT_MAX_ROWS, 'pending') expect(() => @@ -226,7 +228,10 @@ describe('OrchestrationDb dispatch assignee index migration', () => { tempDir = mkdtempSync(join(tmpdir(), 'orca-dispatch-index-migration-')) const dbPath = join(tempDir, 'orchestration.db') db = new OrchestrationDb(dbPath) - const task = db.createTask({ spec: 'indexed lookup' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'indexed lookup' + }) const dispatch = createRootDispatch(db, task.id, 'term_worker') db.close() db = undefined @@ -245,7 +250,9 @@ describe('OrchestrationDb dispatch assignee index migration', () => { db = new OrchestrationDb(dbPath) const sqlite = sqliteFor(db) expect(sqlite.pragma('user_version', { simple: true })).toBe(SCHEMA_VERSION) - expect(db.getDispatchContextById(dispatch.id)).toMatchObject({ assignee_handle: 'term_worker' }) + expect(db.getDispatchContextById(dispatch.id)).toMatchObject({ + assignee_handle: 'term_worker' + }) expect(db.getTask(task.id)).toMatchObject({ created_by_pane_key: null, created_by_process_incarnation: null, diff --git a/src/main/runtime/orchestration/orchestration-federated-legacy-probe.test.ts b/src/main/runtime/orchestration/orchestration-federated-legacy-probe.test.ts new file mode 100644 index 00000000000..5eff7569d27 --- /dev/null +++ b/src/main/runtime/orchestration/orchestration-federated-legacy-probe.test.ts @@ -0,0 +1,140 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { LEGACY_RUN_ID, OrchestrationDb } from './db' +import { federatedStubHomeRunId, SCHEMA_VERSION, UNBOUND_RUN_ID } from './db/contract-constants' +import { resolveOrchestrationMigrationStartVersion } from './orchestration-schema-version-skew' + +describe('federated mailbox legacy-adoption probe', () => { + let db: OrchestrationDb | undefined + let directory: string | undefined + + afterEach(() => { + db?.close() + if (directory) { + rmSync(directory, { recursive: true, force: true }) + } + }) + + function seedMailbox( + handle: string, + kind: 'message' | 'delivery', + homeRunId = 'run_home', + mailRunId = LEGACY_RUN_ID + ): string { + directory = mkdtempSync(join(tmpdir(), 'orca-federated-legacy-probe-')) + const path = join(directory, 'orchestration.db') + db = new OrchestrationDb(path) + db.db + .prepare( + `INSERT INTO remote_dispatch_attachments ( + dispatch_id, task_id, home_peer_fingerprint, home_run_id, runtime_epoch, state + ) VALUES ('ctx_remote', 'task_remote', 'peer_home', ?, 'epoch', 'ready')` + ) + .run(homeRunId) + if (kind === 'message') { + db.db + .prepare( + `INSERT INTO messages ( + id, run_id, delivery_contract, from_handle, to_handle, subject, type + ) VALUES ('msg_probe', ?, 'current_delivery', 'term_home', ?, 'continue', 'dispatch')` + ) + .run(mailRunId, handle) + } else { + db.db + .prepare( + `INSERT INTO deliveries (id, run_id, mailbox_handle, consumer_generation, message_ids) + VALUES ('delivery_probe', ?, ?, 0, '[]')` + ) + .run(mailRunId, handle) + } + return path + } + + it.each(['message', 'delivery'] as const)( + 'does not replay adoption for a misfiled federated %s', + (kind) => { + const path = seedMailbox('dispatch:ctx_remote', kind) + expect( + resolveOrchestrationMigrationStartVersion(db!.db, SCHEMA_VERSION, SCHEMA_VERSION) + ).toBe(SCHEMA_VERSION) + db!.close() + db = new OrchestrationDb(path) + expect(db.getLegacyAdoption()).toBeUndefined() + if (kind === 'message') { + expect(db.getMessageById('msg_probe')).toMatchObject({ + run_id: LEGACY_RUN_ID, + delivery_contract: 'current_delivery' + }) + } else { + expect( + db.db.prepare("SELECT status FROM deliveries WHERE id = 'delivery_probe'").get() + ).toEqual({ + status: 'outstanding' + }) + } + } + ) + + it.each(['message', 'delivery'] as const)( + 'does not treat a stub-home-Run attachment %s as pre-Runs evidence', + (kind) => { + const stubRunId = federatedStubHomeRunId('ctx_remote') + const path = seedMailbox('dispatch:ctx_remote', kind, stubRunId, stubRunId) + db!.db + .prepare( + `INSERT INTO runs (id, objective, home_database, consumer_generation, legacy) + VALUES (?, 'Coordinated from peer_home', 'remote', 0, 0)` + ) + .run(stubRunId) + expect( + resolveOrchestrationMigrationStartVersion(db!.db, SCHEMA_VERSION, SCHEMA_VERSION) + ).toBe(SCHEMA_VERSION) + db!.close() + db = new OrchestrationDb(path) + expect(db.getLegacyAdoption()).toBeUndefined() + expect(db.getRemoteDispatchAttachment('ctx_remote')?.home_run_id).toBe(stubRunId) + } + ) + + it.each(['message', 'delivery'] as const)( + 'still replays adoption for a genuine legacy %s', + (kind) => { + const path = seedMailbox('term_legacy_coordinator', kind) + expect( + resolveOrchestrationMigrationStartVersion(db!.db, SCHEMA_VERSION, SCHEMA_VERSION) + ).toBe(6) + db!.close() + db = new OrchestrationDb(path) + expect(db.getLegacyAdoption()).toBeDefined() + if (kind === 'message') { + expect(db.getMessageById('msg_probe')).toMatchObject({ + run_id: db.getLegacyAdoption()!.adopted_run_id, + delivery_contract: 'legacy_direct' + }) + } else { + expect( + db.db.prepare("SELECT status FROM deliveries WHERE id = 'delivery_probe'").get() + ).toEqual({ + status: 'fenced' + }) + } + } + ) + + it('keeps mail from a terminal in no Run across a reopen without replaying adoption', () => { + directory = mkdtempSync(join(tmpdir(), 'orca-unbound-mail-probe-')) + const path = join(directory, 'orchestration.db') + db = new OrchestrationDb(path) + const sent = db.insertMessage({ from: 'term_a', to: 'term_b', subject: 'hi' }) + expect(sent.run_id).toBe(UNBOUND_RUN_ID) + expect(resolveOrchestrationMigrationStartVersion(db.db, SCHEMA_VERSION, SCHEMA_VERSION)).toBe( + SCHEMA_VERSION + ) + db.close() + db = new OrchestrationDb(path) + expect(db.getLegacyAdoption()).toBeUndefined() + expect(db.getUnreadMessages('term_b').map((row) => row.id)).toEqual([sent.id]) + }) +}) diff --git a/src/main/runtime/orchestration/orchestration-legacy-storage-test-fixture.ts b/src/main/runtime/orchestration/orchestration-legacy-storage-test-fixture.ts index 4cdc8f5ee91..886f2383db5 100644 --- a/src/main/runtime/orchestration/orchestration-legacy-storage-test-fixture.ts +++ b/src/main/runtime/orchestration/orchestration-legacy-storage-test-fixture.ts @@ -54,6 +54,7 @@ export function createLegacyStorageCutoverFixture(): { }) const legacyTask = first.createTask({ + runId: 'run_legacy_local', spec: 'legacy', createdByTerminalHandle: 'term_legacy_coord' }) @@ -76,16 +77,19 @@ export function createLegacyStorageCutoverFixture(): { ) const legacyMessages = [ first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_coord', to: 'term_legacy_worker', subject: 'read worker mail' }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'read coordinator mail' }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_coord', to: 'term_legacy_worker', subject: 'second worker page' @@ -99,6 +103,7 @@ export function createLegacyStorageCutoverFixture(): { question: 'Retained question?' }) const rejection = first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Rejected heartbeat', @@ -106,6 +111,7 @@ export function createLegacyStorageCutoverFixture(): { payload: JSON.stringify({ _orcaLifecycleRejection: { code: 'migration', reason: 'cutover' } }) }) const lookalike = first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Ordinary legacy mail', @@ -115,36 +121,42 @@ export function createLegacyStorageCutoverFixture(): { }) const malformedRejections = [ first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Invalid JSON marker', payload: '{"_orcaLifecycleRejection":' }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Array marker', payload: JSON.stringify({ _orcaLifecycleRejection: [] }) }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'String marker', payload: JSON.stringify({ _orcaLifecycleRejection: 'migration' }) }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Incomplete marker', payload: JSON.stringify({ _orcaLifecycleRejection: { code: 'migration' } }) }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Non-string marker fields', payload: JSON.stringify({ _orcaLifecycleRejection: { code: 19, reason: false } }) }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'Array root', @@ -153,6 +165,7 @@ export function createLegacyStorageCutoverFixture(): { ]) }), first.insertMessage({ + runId: 'run_legacy_local', from: 'term_legacy_worker', to: 'term_legacy_coord', subject: 'String root', diff --git a/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts b/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts index abb0b7bdfcc..7d0f0931263 100644 --- a/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts +++ b/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts @@ -26,14 +26,6 @@ function recoveryRow( describe('legacy worker terminal recovery planning', () => { it('retains completed Dispatches when the worker process row is still live', () => { expect(planLegacyWorkerTerminalRecovery([recoveryRow()])).toEqual({ - blockedPanes: [ - { - worktreeId: 'repo::/workspace', - paneKey: `tab-worker:${LEAF_ID}`, - contractVersion: 0, - settled: false - } - ], candidates: [ expect.objectContaining({ dispatchId: 'dispatch-1', @@ -45,18 +37,10 @@ describe('legacy worker terminal recovery planning', () => { }) }) - it('blocks resume but refuses recovery when durable handles disagree', () => { + it('refuses recovery when durable handles disagree', () => { expect( planLegacyWorkerTerminalRecovery([recoveryRow({ agent_terminal_handle: 'term-replacement' })]) ).toEqual({ - blockedPanes: [ - { - worktreeId: 'repo::/workspace', - paneKey: `tab-worker:${LEAF_ID}`, - contractVersion: 0, - settled: false - } - ], candidates: [], ambiguousDispatchIds: [] }) @@ -70,8 +54,6 @@ describe('legacy worker terminal recovery planning', () => { expect(plan.candidates).toEqual([expect.objectContaining({ dispatchId: 'dispatch-live' })]) expect(plan.ambiguousDispatchIds).toEqual([]) - // A live dispatch still holds this pane, so it must not be reported as a settled fence. - expect(plan.blockedPanes).toEqual([expect.objectContaining({ settled: false })]) }) it('fails closed when two Dispatches claim one terminal identity', () => { @@ -82,7 +64,6 @@ describe('legacy worker terminal recovery planning', () => { expect(plan.candidates).toEqual([]) expect(plan.ambiguousDispatchIds).toEqual(['dispatch-1', 'dispatch-2']) - expect(plan.blockedPanes).toHaveLength(1) }) it('does not trust malformed pane or process identities', () => { @@ -94,7 +75,6 @@ describe('legacy worker terminal recovery planning', () => { ]) expect(plan).toEqual({ - blockedPanes: [], candidates: [], ambiguousDispatchIds: [] }) diff --git a/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.ts b/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.ts index 8b1426cb07e..7a159a89ea0 100644 --- a/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.ts +++ b/src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.ts @@ -19,16 +19,7 @@ export type LegacyWorkerTerminalRecoveryCandidate = { incarnationId: PtyIncarnationId } -export type LegacyWorkerTerminalRecoveryBlockedPane = { - worktreeId: string - paneKey: string - contractVersion: number - /** The dispatch reported an outcome; its pane needs the fence but owns no process to recover. */ - settled: boolean -} - export type LegacyWorkerTerminalRecoveryPlan = { - blockedPanes: LegacyWorkerTerminalRecoveryBlockedPane[] candidates: LegacyWorkerTerminalRecoveryCandidate[] ambiguousDispatchIds: string[] } @@ -65,26 +56,13 @@ function countCandidateKeys( export function planLegacyWorkerTerminalRecovery( rows: readonly LegacyWorkerTerminalRecoveryRow[] ): LegacyWorkerTerminalRecoveryPlan { - const blockedPanes = new Map() const parsedCandidates: LegacyWorkerTerminalRecoveryCandidate[] = [] for (const row of rows) { const worktreeId = row.worktree_id?.trim() const paneKey = row.assignee_pane_key?.trim() const pane = paneKey ? parsePaneKey(paneKey) : null const settled = WORKER_SETTLED_STATES.includes(row.worker_state) - if (worktreeId && paneKey && pane) { - const blockedKey = `${worktreeId}\0${paneKey}` - const alreadySettled = blockedPanes.get(blockedKey)?.settled - blockedPanes.set(blockedKey, { - worktreeId, - paneKey, - contractVersion: row.contract_version, - // A pane reused across dispatches is settled only once every dispatch holding it is. - settled: (alreadySettled ?? true) && settled - }) - } - // A settled worker owns no live process to adopt or roll back, so its identity must never - // compete with a running worker's in the ambiguity count below. + // Settled dispatches need no adoption and must not make an active worker's identity ambiguous. if (settled) { continue } @@ -134,7 +112,6 @@ export function planLegacyWorkerTerminalRecovery( return !ambiguous }) return { - blockedPanes: [...blockedPanes.values()], candidates, ambiguousDispatchIds: [...ambiguousDispatchIds] } diff --git a/src/main/runtime/orchestration/orchestration-mutation-question-db.test.ts b/src/main/runtime/orchestration/orchestration-mutation-question-db.test.ts index 5a26448bd98..c4911b7d676 100644 --- a/src/main/runtime/orchestration/orchestration-mutation-question-db.test.ts +++ b/src/main/runtime/orchestration/orchestration-mutation-question-db.test.ts @@ -77,6 +77,7 @@ describe('OrchestrationDb mutation and question state', () => { it('accepts a question message in the fresh canonical schema', () => { const d = createDb() const message = d.insertMessage({ + runId: 'run_legacy_local', from: 'worker', to: 'run:run_1', subject: 'Need input', diff --git a/src/main/runtime/orchestration/orchestration-schema-version-skew.ts b/src/main/runtime/orchestration/orchestration-schema-version-skew.ts index a2767e1e5a7..85e0a78b3ae 100644 --- a/src/main/runtime/orchestration/orchestration-schema-version-skew.ts +++ b/src/main/runtime/orchestration/orchestration-schema-version-skew.ts @@ -42,7 +42,8 @@ const VERSIONED_POST_V6_COLUMNS = [ { version: 36, table: 'dispatch_contexts', column: 'consumer_generation' }, { version: 36, table: 'remote_dispatch_attachments', column: 'consumer_generation' }, { version: 37, table: 'dispatch_contexts', column: 'creator_handle' }, - { version: 37, table: 'dispatch_contexts', column: 'creator_pane_key' } + { version: 37, table: 'dispatch_contexts', column: 'creator_pane_key' }, + { version: 40, table: 'remote_dispatch_attachments', column: 'home_run_id' } ] as const // Why: v34 shipped without these two, so a v34 stamp proves nothing about them; v35 repairs both @@ -122,15 +123,22 @@ function messagesAllowQuestions(db: Database.Database): boolean { function hasConsistentLegacyAdoption(db: Database.Database): boolean { const sourceRunId = 'run_legacy_local' + // Misfiled federated mail is not evidence of a pre-Runs database. + const notFederatedMailbox = (handle: string): string => + `NOT EXISTS (SELECT 1 FROM remote_dispatch_attachments AS attachment + WHERE 'dispatch:' || attachment.dispatch_id = ${handle})` + const deliveryFilter = hasOrchestrationColumn(db, 'deliveries', 'mailbox_handle') + ? ` AND ${notFederatedMailbox('mailbox_handle')}` + : '' const sourceGraph = db .prepare( `SELECT 1 WHERE EXISTS(SELECT 1 FROM tasks WHERE run_id = ?) OR EXISTS(SELECT 1 FROM dispatch_contexts WHERE run_id = ?) OR EXISTS(SELECT 1 FROM decision_gates WHERE run_id = ?) - OR EXISTS(SELECT 1 FROM messages WHERE run_id = ?) + OR EXISTS(SELECT 1 FROM messages WHERE run_id = ? AND ${notFederatedMailbox('to_handle')}) OR EXISTS(SELECT 1 FROM question_threads WHERE run_id = ?) - OR EXISTS(SELECT 1 FROM deliveries WHERE run_id = ?)` + OR EXISTS(SELECT 1 FROM deliveries WHERE run_id = ?${deliveryFilter})` ) .get(sourceRunId, sourceRunId, sourceRunId, sourceRunId, sourceRunId, sourceRunId) const adoption = db diff --git a/src/main/runtime/orchestration/orchestration-settled-worker-resume-fence-db.test.ts b/src/main/runtime/orchestration/orchestration-settled-worker-resume-fence-db.test.ts deleted file mode 100644 index ee52bc026d0..00000000000 --- a/src/main/runtime/orchestration/orchestration-settled-worker-resume-fence-db.test.ts +++ /dev/null @@ -1,124 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import { OrchestrationDb } from './db' -import { planLegacyWorkerTerminalRecovery } from './orchestration-legacy-worker-terminal-recovery' -import type { WorkerTerminalResourceRow } from './worker-terminal-ownership' - -const PANE_KEY = 'tab_worker:33333333-3333-4333-8333-333333333333' - -describe('settled worker terminal resume fence rows', () => { - let db: OrchestrationDb | undefined - - afterEach(() => db?.close()) - - function createReadyWorker(): { db: OrchestrationDb; taskId: string; dispatchId: string } { - const d = new OrchestrationDb(':memory:') - db = d - const task = d.createTask({ spec: 'settled worker' }) - const started = d.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: task.id, - startOptions: {} - }) - d.prepareStartingWorkerAuthority({ - dispatchId: started.dispatch.id, - handle: 'term_worker', - paneKey: PANE_KEY, - processIncarnation: 'runtime:pty:1', - worktreeId: 'repo::worktree', - setupState: 'not_applicable', - effects: [], - terminalOwnership: 'created' - }) - d.markWorkerDispatchReady(started.dispatch.id) - return { db: d, taskId: task.id, dispatchId: started.dispatch.id } - } - - /** Asserts the `requested` arm so the resource row is non-null for the caller. */ - function requestRelease(d: OrchestrationDb, dispatchId: string): WorkerTerminalResourceRow { - const requested = d.requestWorkerTerminalRelease(dispatchId) - if (requested.disposition !== 'requested') { - throw new Error(`expected a release request, got ${requested.disposition}`) - } - return requested.resource - } - - function settle(d: OrchestrationDb, taskId: string, dispatchId: string): void { - expect( - d.settleWorkerReport({ - taskId, - dispatchId, - outcome: 'succeeded', - result: 'worker succeeded' - }).action - ).toBe('settled') - } - - it('keeps a settled-but-unreleased worker terminal in the recovery rows', () => { - const { db: d, taskId, dispatchId } = createReadyWorker() - settle(d, taskId, dispatchId) - - expect(d.getWorkerDispatch(dispatchId)?.state).toBe('succeeded') - expect(d.listLegacyWorkerTerminalRecoveryRows()).toEqual([ - expect.objectContaining({ - dispatch_id: dispatchId, - worker_state: 'succeeded', - assignee_pane_key: PANE_KEY - }) - ]) - }) - - // A settled worker owns no live process, so it must only fence — never be offered for adoption. - it('plans a settled pane as a fence with no adoption candidate', () => { - const { db: d, taskId, dispatchId } = createReadyWorker() - settle(d, taskId, dispatchId) - - const plan = planLegacyWorkerTerminalRecovery(d.listLegacyWorkerTerminalRecoveryRows()) - - expect(plan.blockedPanes).toEqual([ - expect.objectContaining({ paneKey: PANE_KEY, settled: true }) - ]) - expect(plan.candidates).toEqual([]) - expect(plan.ambiguousDispatchIds).toEqual([]) - }) - - // `release_unknown` is the ticket's own repro: release could not be proven, the pane keeps a - // resumable provider session, and dropping it here would re-open the auto-resume. - it('keeps a settled worker terminal whose release could not be proven', () => { - const { db: d, taskId, dispatchId } = createReadyWorker() - settle(d, taskId, dispatchId) - const resource = requestRelease(d, dispatchId) - expect( - d.markWorkerTerminalReleaseUnknown(resource.id, 'terminal no longer resolves').release_state - ).toBe('unknown') - - expect(d.listLegacyWorkerTerminalRecoveryRows()).toEqual([ - expect.objectContaining({ dispatch_id: dispatchId, assignee_pane_key: PANE_KEY }) - ]) - }) - - it('drops a settled worker terminal once its resource is released', () => { - const { db: d, taskId, dispatchId } = createReadyWorker() - settle(d, taskId, dispatchId) - const resource = requestRelease(d, dispatchId) - expect(d.settleWorkerTerminalRelease(resource.id).release_state).toBe('released') - - expect(d.listLegacyWorkerTerminalRecoveryRows()).toEqual([]) - }) - - it('drops a settled worker terminal the user chose to retain', () => { - const { db: d, taskId, dispatchId } = createReadyWorker() - d.retainWorkerTerminalResource(dispatchId) - settle(d, taskId, dispatchId) - - expect(d.listLegacyWorkerTerminalRecoveryRows()).toEqual([]) - }) - - it('drops a settled worker terminal the user took over', () => { - const { db: d, taskId, dispatchId } = createReadyWorker() - settle(d, taskId, dispatchId) - expect(d.markWorkerTerminalUserOwned(PANE_KEY)).toBe(1) - - expect(d.listLegacyWorkerTerminalRecoveryRows()).toEqual([]) - }) -}) diff --git a/src/main/runtime/orchestration/orchestration-version-skew-migration.test.ts b/src/main/runtime/orchestration/orchestration-version-skew-migration.test.ts index 12ccf7303ca..f8fa7a5df48 100644 --- a/src/main/runtime/orchestration/orchestration-version-skew-migration.test.ts +++ b/src/main/runtime/orchestration/orchestration-version-skew-migration.test.ts @@ -389,7 +389,10 @@ describe('OrchestrationDb version-skew migration', () => { tempDir = mkdtempSync(join(tmpdir(), 'orca-db-version-skew-v30-reset-')) const dbPath = join(tempDir, 'orchestration.db') db = new OrchestrationDb(dbPath) - const task = db.createTask({ spec: 'reset by an older writer' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'reset by an older writer' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts b/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts index 16a118008de..d6e6038cd65 100644 --- a/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts +++ b/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts @@ -15,7 +15,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('creates and activates a composed worker Dispatch transactionally', () => { const d = createDb() - const task = d.createTask({ spec: 'worker' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'worker' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -82,7 +82,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('retains an active supervised worker terminal', () => { const d = createDb() - const task = d.createTask({ spec: 'retain active worker' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'retain active worker' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -114,7 +114,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('requeues an active Task before settling a worker whose terminal is missing', () => { const d = createDb() - const task = d.createTask({ spec: 'recover missing worker' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'recover missing worker' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -153,7 +153,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('commits worker-start mutation acceptance with the starting Dispatch', () => { const d = createDb() - const task = d.createTask({ spec: 'atomic acceptance' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'atomic acceptance' }) const mutationReceipt = { callerFingerprint: 'caller_fingerprint', requestId: 'worker_start_request', @@ -207,7 +207,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('fails a composed start without losing residual resource receipts', () => { const d = createDb() - const task = d.createTask({ spec: 'worker' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'worker' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -232,7 +232,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('allows retry only from the Task current terminal Dispatch', () => { const d = createDb() - const task = d.createTask({ spec: 'retry current' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'retry current' }) const first = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -272,7 +272,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('treats abandon of a superseded Dispatch as a no-op', () => { const d = createDb() - const task = d.createTask({ spec: 'stale abandon' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'stale abandon' }) const first = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -317,7 +317,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('lets the stop fence win before a late worker completion', () => { const d = createDb() - const task = d.createTask({ spec: 'race' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'race' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -350,7 +350,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('allows explicit stop recovery from uncertain local and remote starts', () => { const d = createDb() - const task = d.createTask({ spec: 'uncertain local start' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'uncertain local start' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, @@ -365,6 +365,7 @@ describe('OrchestrationDb worker Dispatch state', () => { }) d.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId: 'ctx_remote_unknown', taskId: 'task_remote_unknown', homePeerFingerprint: 'home_peer', @@ -398,6 +399,7 @@ describe('OrchestrationDb worker Dispatch state', () => { const paneKey = 'tab_remote:11111111-1111-4111-8111-111111111111' const attach = (dispatchId: string): void => { d.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: `task_${dispatchId}`, homePeerFingerprint: 'home_peer', @@ -452,6 +454,7 @@ describe('OrchestrationDb worker Dispatch state', () => { const leafId = '11111111-1111-4111-8111-111111111111' const attach = (dispatchId: string, paneKey: string): void => { d.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: `task_${dispatchId}`, homePeerFingerprint: 'home_peer', @@ -499,7 +502,7 @@ describe('OrchestrationDb worker Dispatch state', () => { it('returns already-settled when completion wins before stop', () => { const d = createDb() - const task = d.createTask({ spec: 'race' }) + const task = d.createTask({ runId: 'run_legacy_local', spec: 'race' }) const started = d.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/r1-identity-migration.test.ts b/src/main/runtime/orchestration/r1-identity-migration.test.ts index bb263d0b9ce..673a72fd844 100644 --- a/src/main/runtime/orchestration/r1-identity-migration.test.ts +++ b/src/main/runtime/orchestration/r1-identity-migration.test.ts @@ -27,7 +27,7 @@ describe('R1 identity migration', () => { tempDir = mkdtempSync(join(tmpdir(), 'orca-r1-identity-')) const dbPath = join(tempDir, 'orchestration.db') db = new OrchestrationDb(dbPath) - const task = db.createTask({ spec: 'legacy supervised worker' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'legacy supervised worker' }) const started = db.createStartingWorkerDispatch({ taskId: task.id, startOptions: { worktree: 'folder:/workspace' }, diff --git a/src/main/runtime/orchestration/types.ts b/src/main/runtime/orchestration/types.ts index 00005443006..85d5dcfc159 100644 --- a/src/main/runtime/orchestration/types.ts +++ b/src/main/runtime/orchestration/types.ts @@ -190,6 +190,7 @@ export type FederatedDispatchRow = { } export type RemoteDispatchAttachmentRow = { + home_run_id: string dispatch_id: string task_id: string home_peer_fingerprint: string diff --git a/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts b/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts index 382ec304bb6..1d123f51b38 100644 --- a/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts +++ b/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts @@ -6,7 +6,7 @@ const INCARNATION = 'runtime_test:term_worker:1' let db: OrchestrationDb function startWorker(spec: string): { taskId: string; dispatchId: string; capability: string } { - const task = db.createTask({ spec }) + const task = db.createTask({ runId: 'run_legacy_local', spec }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/orchestration/worker-transcript-payload.test.ts b/src/main/runtime/orchestration/worker-transcript-payload.test.ts index f47a46605e1..aaa3fdcfcc1 100644 --- a/src/main/runtime/orchestration/worker-transcript-payload.test.ts +++ b/src/main/runtime/orchestration/worker-transcript-payload.test.ts @@ -146,6 +146,36 @@ describe('worker transcript wire bounds', () => { expect(result).toMatchObject({ limited: false, warnings: [] }) }) + it('keeps two roster ids sharing a 512-char prefix distinct', () => { + // The id is the roster key: a plain prefix clip would merge the two children. + const head = 'a'.repeat(512) + const result = boundWorkerTranscriptMessages([ + { + id: 'message-1', + role: 'assistant', + timestamp: null, + source: 'transcript', + blocks: [ + { + type: 'subagent-group', + groupId: 'g', + agents: [ + { id: `${head}-one`, label: 'Audit', state: 'working' }, + { id: `${head}-two`, label: 'Audit', state: 'working' } + ] + } + ] + } + ]) + + const block = result.messages[0]?.blocks[0] + if (block?.type !== 'subagent-group') { + throw new Error('expected a subagent-group block') + } + expect(block.agents[0]?.id).not.toBe(block.agents[1]?.id) + expect(block.agents[0]?.id).toHaveLength(512) + }) + it('keeps fallback identifiers stable without exposing the transcript path', () => { const transcriptPath = 'C:\\Users\\worker\\.codex\\session.jsonl' const message = { diff --git a/src/main/runtime/orchestration/worker-transcript-payload.ts b/src/main/runtime/orchestration/worker-transcript-payload.ts index f9d83e20c62..0f50aa8309c 100644 --- a/src/main/runtime/orchestration/worker-transcript-payload.ts +++ b/src/main/runtime/orchestration/worker-transcript-payload.ts @@ -5,6 +5,7 @@ import type { NativeChatMessage, NativeChatSubagentState } from '../../../shared/native-chat-types' +import { boundSubagentEntryId } from '../../native-chat/subagent-entry-id-bounds' export const DEFAULT_WORKER_TRANSCRIPT_MESSAGE_LIMIT = 40 export const MAX_WORKER_TRANSCRIPT_MESSAGE_LIMIT = 50 @@ -153,7 +154,7 @@ function boundBlock(block: NativeChatBlock, state: TranscriptBoundState): Native groupId: clipMetadata(block.groupId, state), agents: agents.map((agent) => ({ ...agent, - id: clipMetadata(agent.id, state), + id: boundEntryId(agent.id, state), label: clipMetadata(agent.label, state), state: clipSubagentState(agent.state, state) })) @@ -194,6 +195,18 @@ function isLocalFileLocator(value: string): boolean { ) } +/** A roster entry's id is the roster KEY, so it is redacted like other metadata + * but bounded with a digest rather than clipped: two ids sharing a 512-char + * head must not collapse onto one entry. */ +function boundEntryId(value: string, state: TranscriptBoundState): string { + const redacted = redactSensitiveText(value, state.warnings) + const bounded = boundSubagentEntryId(redacted) + if (bounded !== redacted) { + markClipped(state, 'Oversized transcript metadata was clipped.') + } + return bounded +} + function clipMetadata(value: string, state: TranscriptBoundState): string { const redacted = redactSensitiveText(value, state.warnings) if (redacted.length <= MAX_WORKER_TRANSCRIPT_METADATA_CHARS) { diff --git a/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.test.ts b/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.test.ts new file mode 100644 index 00000000000..de25ea7e45d --- /dev/null +++ b/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it } from 'vitest' +import type { NativeChatBlock } from '../../../../shared/native-chat-types' +import { sanitizeNativeChatRpcBlock } from './native-chat-rpc-block-sanitize' + +const SHARED_HEAD = 'a'.repeat(512) + +function rosterBlock(ids: readonly string[]): NativeChatBlock { + return { + type: 'subagent-group', + groupId: 'group-1', + agents: ids.map((id) => ({ id, label: 'l'.repeat(900), state: 'working' as const })) + } +} + +describe('mobile subagent roster bounds', () => { + it('keeps two ids sharing a 512-char prefix distinct', () => { + const block = sanitizeNativeChatRpcBlock( + rosterBlock([`${SHARED_HEAD}-one`, `${SHARED_HEAD}-two`]), + 'mobile' + ) + + if (block.type !== 'subagent-group') { + throw new Error('expected a subagent-group block') + } + expect(block.agents[0]?.id).not.toBe(block.agents[1]?.id) + expect(block.agents[0]?.id).toHaveLength(512) + // The label is display text and still clips. + expect(block.agents[0]?.label).toContain('… (truncated)') + }) + + it('leaves a short id alone', () => { + const block = sanitizeNativeChatRpcBlock(rosterBlock(['task-1']), 'mobile') + expect(block.type === 'subagent-group' && block.agents[0]?.id).toBe('task-1') + }) +}) diff --git a/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.ts b/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.ts index fc19273d5ca..79f1da4fc53 100644 --- a/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.ts +++ b/src/main/runtime/rpc/methods/native-chat-rpc-block-sanitize.ts @@ -3,6 +3,7 @@ import { normalizeSubagentState } from '../../../../shared/native-chat-subagent-summary' import type { NativeChatBlock, NativeChatSubagentState } from '../../../../shared/native-chat-types' +import { boundSubagentEntryId } from '../../../native-chat/subagent-entry-id-bounds' import type { RpcContext } from '../core' import { sanitizeNativeChatRpcImageBlock } from './native-chat-rpc-image-block' @@ -63,7 +64,9 @@ export function sanitizeNativeChatRpcBlock( groupId: clip(block.groupId, MAX_SUBAGENT_FIELD_CHARS), agents: block.agents.slice(0, MOBILE_SUBAGENT_CAP).map((agent) => ({ ...agent, - id: clip(agent.id, MAX_SUBAGENT_FIELD_CHARS), + // The id is the roster KEY: a prefix clip would merge two children, so + // it takes the shared digest bound the other wires use. + id: boundSubagentEntryId(agent.id), label: clip(agent.label, MAX_SUBAGENT_FIELD_CHARS), state: clipSubagentState(agent.state) })) diff --git a/src/main/runtime/rpc/methods/native-chat.test.ts b/src/main/runtime/rpc/methods/native-chat.test.ts index 1417e716770..47a6d9e855e 100644 --- a/src/main/runtime/rpc/methods/native-chat.test.ts +++ b/src/main/runtime/rpc/methods/native-chat.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it, vi } from 'vitest' -import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import type { + NativeChatMessage, + NativeChatSubagentEntry +} from '../../../../shared/native-chat-types' import type { RpcContext } from '../core' // Stub the bounded tail reader so the handler returns a deterministic transcript with @@ -88,6 +91,7 @@ vi.mock('../../../native-chat/transcript-watch', () => ({ } })) +import { boundSubagentEntryId } from '../../../native-chat/subagent-entry-id-bounds' import { NATIVE_CHAT_METHODS } from './native-chat' function makeMessage(text: string): NativeChatMessage { @@ -191,6 +195,35 @@ describe('nativeChat.readSession clientKind truncation gating', () => { expect(block.text).toBe(text) }) + it('bounds a subagent roster before it reaches mobile', async () => { + const agents: NativeChatSubagentEntry[] = Array.from({ length: 100 }, (_, index) => ({ + id: `task-${index}-${'i'.repeat(600)}`, + label: 'l'.repeat(600), + state: 'working' + })) + cachedResult.value = { + messages: [ + { + ...makeMessage(''), + blocks: [{ type: 'subagent-group', groupId: 'g', agents }] + } + ] + } + const result = await readSessionHandler()( + { agent: 'claude', sessionId: 's' }, + ctxWith('mobile') + ) + const block = (result as { messages: NativeChatMessage[] }).messages[0].blocks[0] as { + agents: { id: string; label: string }[] + } + expect(block.agents).toHaveLength(64) + expect(block.agents[0].label).toBe(`${'l'.repeat(512)}\n… (truncated)`) + // The id is as untrusted as the label on an imported roster, but it is the + // roster key: it is bounded with a digest, never clipped to a bare prefix. + expect(block.agents[0].id).toHaveLength(512) + expect(block.agents[0].id).toBe(boundSubagentEntryId(`task-0-${'i'.repeat(600)}`)) + }) + it('clips a pathological text block at the safety ceiling for mobile clients', async () => { const text = 'y'.repeat(70_000) cachedResult.value = { messages: [makeTextMessage(text)] } diff --git a/src/main/runtime/rpc/methods/orchestration-structured-worker-session.ts b/src/main/runtime/rpc/methods/orchestration-structured-worker-session.ts index 9ed0c05ad1c..6034a72601f 100644 --- a/src/main/runtime/rpc/methods/orchestration-structured-worker-session.ts +++ b/src/main/runtime/rpc/methods/orchestration-structured-worker-session.ts @@ -80,6 +80,8 @@ export async function createStructuredWorkerSession(args: { worktreeId: string agent: 'claude' | 'codex' dispatchId: string + /** The dispatch's own `--model`/`--effort`, already narrowed to the seedable string subset. */ + options?: Readonly> /** Retried whenever the session's journal moves, which is the structured idle edge. */ onJournalActivity: (sessionId: string) => void }): Promise<{ identity: StructuredWorkerIdentity; host: StructuredAgentSessionHost }> { @@ -120,6 +122,8 @@ export async function createStructuredWorkerSession(args: { }, worktree: `id:${args.worktreeId}`, agent: args.agent, + // Absent, the host seeds the user's saved selection — the same fallback a chat gets. + ...(args.options ? { options: args.options } : {}), // Dispatching a worker is background work; it must not pull the surface away from the user. activate: false }) diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-mode-selection.test.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-mode-selection.test.ts index e87f8182037..52cf3579016 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-mode-selection.test.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-mode-selection.test.ts @@ -125,6 +125,29 @@ describe('worker-start honours the settings default', () => { } } + function mockWorktreeCreation() { + vi.spyOn(runtime, 'showManagedWorktree').mockResolvedValue({ + id: 'repo::wt', + repoId: 'repo' + } as never) + vi.spyOn(runtime, 'showRepo').mockResolvedValue({ id: 'repo', kind: 'git' } as never) + vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ terminals: [] } as never) + return vi.spyOn(runtime, 'createManagedWorktree').mockImplementation( + async (createArgs) => + ({ + worktree: { id: 'repo::child', repoId: 'repo' }, + ...(createArgs.startupAgent + ? { startupTerminal: { spawned: true, handle: TERMINAL_HANDLE } } + : {}), + setupReceipt: { + hookFound: false, + startupPolicy: 'start-immediately', + state: 'not_configured' + } + }) as never + ) + } + it('starts a structured chat worker when structured native chat is the default', async () => { const result = await startWorker(STRUCTURED_DEFAULT) @@ -172,12 +195,74 @@ describe('worker-start honours the settings default', () => { expect(createExistingWorktreeWorkerTerminal).toHaveBeenCalledTimes(1) }) - it('falls back instead of refusing a launch preference the structured default cannot apply', async () => { + it('seeds --model and --effort into the structured session instead of downgrading', async () => { + // These two used to force a PTY worker, which is half of why orchestration never produced a + // structured chat: choosing a model is the ordinary way to dispatch one. const result = await startWorker(STRUCTURED_DEFAULT, { model: 'opus', effort: 'high' }) expect(result).toMatchObject({ state: 'ready', - mode: { mode: 'terminal', preferred: 'structured', reason: 'launch_preferences' } + mode: { mode: 'structured', preferred: 'structured', reason: 'user_default' } + }) + expect(createExistingWorktreeWorkerTerminal).not.toHaveBeenCalled() + expect(createStructuredWorkerSessionForWorktree).toHaveBeenCalledWith( + expect.objectContaining({ launchPreferences: { model: 'opus', effort: 'high' } }) + ) + }) + + it('creates a new worktree WITHOUT an agent terminal and gives it a structured session', async () => { + // The other half: `createWorkerWorktree` creates agent-first, so a `--worktree new-child` + // dispatch could only ever end up a PTY terminal worker. + const created = mockWorktreeCreation() + + const result = await startWorker(STRUCTURED_DEFAULT, { + worktree: 'new-child', + name: 'worker-child' + }) + + expect(result).toMatchObject({ + state: 'ready', + mode: { mode: 'structured', preferred: 'structured', reason: 'user_default' } + }) + expect(created).toHaveBeenCalledWith( + expect.not.objectContaining({ startupAgent: expect.anything() }) + ) + expect(createStructuredWorkerSessionForWorktree).toHaveBeenCalledWith( + expect.objectContaining({ worktreeId: 'repo::child' }) + ) + expect(createExistingWorktreeWorkerTerminal).not.toHaveBeenCalled() + }) + + it('still creates the new worktree agent-first when the default is a terminal worker', async () => { + const created = mockWorktreeCreation() + + const result = await startWorker( + { ...STRUCTURED_DEFAULT, experimentalStructuredNativeChat: false }, + { worktree: 'new-child', name: 'worker-child' } + ) + + expect(result).toMatchObject({ state: 'ready', mode: { mode: 'terminal' } }) + expect(created).toHaveBeenCalledWith(expect.objectContaining({ startupAgent: 'claude' })) + expect(createStructuredWorkerSessionForWorktree).not.toHaveBeenCalled() + }) + + it('falls back to a terminal agent in the worktree it just created when the host refuses', async () => { + // The host can only answer for a workspace that exists, so a created worktree settles its mode + // after creation — and a refusal must not fail a routine dispatch. + mockWorktreeCreation() + vi.mocked(runtime.getStructuredAgentSessionCreateSupport).mockResolvedValue({ + supported: false, + reason: 'wsl' + }) + + const result = await startWorker(STRUCTURED_DEFAULT, { + worktree: 'new-child', + name: 'worker-child' + }) + + expect(result).toMatchObject({ + state: 'ready', + mode: { mode: 'terminal', preferred: 'structured', reason: 'wsl_execution_runtime' } }) expect(createExistingWorktreeWorkerTerminal).toHaveBeenCalledTimes(1) expect(createStructuredWorkerSessionForWorktree).not.toHaveBeenCalled() diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.test.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.test.ts index 23ed5b5a549..b937f6febd2 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.test.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.test.ts @@ -63,10 +63,6 @@ describe('a structured default this dispatch cannot honour', () => { it.each([ ['a remote --on', { on: 'server-1' }, 'remote_execution_host'], ['an existing --terminal', { terminal: 'term_1' }, 'reused_terminal'], - ['a new-child worktree', { worktree: 'new-child' }, 'worktree_creation'], - ['a new-top-level worktree', { worktree: 'new-top-level' }, 'worktree_creation'], - ['--model', { model: 'opus' }, 'launch_preferences'], - ['--effort', { effort: 'high' }, 'launch_preferences'], ['a non-structured agent', { agent: 'cursor' }, 'agent_without_structured_session'], ['no agent at all', { agent: undefined }, 'agent_without_structured_session'] ])('falls back to a terminal worker for %s', (_name, params, reason) => { @@ -76,6 +72,22 @@ describe('a structured default this dispatch cannot honour', () => { expect(receipt.detail).toContain('Your default is a structured chat session') }) + it.each([ + ['a new-child worktree', { worktree: 'new-child' }], + ['a new-top-level worktree', { worktree: 'new-top-level' }], + ['--model', { model: 'opus' }], + ['--effort with its model', { model: 'opus', effort: 'high' }], + ['both at once', { worktree: 'new-child', model: 'opus', effort: 'high' }] + ])('no longer downgrades for %s, which is what made structured unreachable', (_name, params) => { + // Both flags are what a routine dispatch passes, and each used to force a PTY worker, so + // orchestration never produced a structured chat in practice. + expect(decide({ params: { agent: 'claude', ...params } })).toMatchObject({ + mode: 'structured', + preferred: 'structured', + reason: 'user_default' + }) + }) + it('keeps the current worktree structured, which is the ordinary dispatch', () => { expect(decide({ params: { agent: 'codex', worktree: 'current' } }).mode).toBe('structured') }) diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts index d079b48de25..97c64082bf2 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts @@ -9,8 +9,8 @@ * * The settings default and the per-launch feasibility both come from * `shared/structured-native-chat-launch-route`, the same module the renderer's - * `resolveAgentLaunchRoute` uses; only the placement options that exist solely on this command are - * decided here. + * `resolveAgentLaunchRoute` uses. This adapter supplies placement facts and formats the receipt; + * it does not own a second feasibility policy. */ import type { GlobalSettings } from '../../../../shared/global-settings-types' @@ -31,11 +31,10 @@ export type WorkerStartModeReason = | 'user_default' | 'remote_execution_host' | 'reused_terminal' - | 'worktree_creation' - | 'launch_preferences' | 'agent_without_structured_session' | 'tui_launch_customization' | 'structured_sessions_unavailable' + | 'structured_support_unknown' | 'wsl_execution_runtime' | 'codex_on_windows' | 'structured_unsupported_on_host' @@ -55,6 +54,9 @@ type WorkerStartModeSettings = Partial< Pick > +/** The placement options that exist only on `worker-start`. `worktree`, `model` and `effort` are + * listed but no longer read: a structured worker honours all three, and naming them here keeps + * the set of options this decision has considered visible. */ type WorkerStartModePlacement = { agent?: string on?: string @@ -65,14 +67,13 @@ type WorkerStartModePlacement = { } const DOWNGRADE_DETAIL: Record, string> = { - remote_execution_host: '--on runs the worker on a remote execution host', + remote_execution_host: 'this worker runs on a remote execution host', reused_terminal: '--terminal reuses a running terminal agent', - worktree_creation: 'a new worktree is created with its agent terminal', - launch_preferences: '--model and --effort apply only to a terminal agent', agent_without_structured_session: 'this agent has no structured session', tui_launch_customization: 'this agent has a custom launch command, arguments or environment that only a terminal applies', structured_sessions_unavailable: 'this runtime does not support structured agent sessions', + structured_support_unknown: 'the execution host has not established structured session support', wsl_execution_runtime: 'this workspace runs under WSL', codex_on_windows: 'Codex has no structured session on Windows', structured_unsupported_on_host: 'the execution host cannot create one here' @@ -82,6 +83,7 @@ const BLOCKER_REASON: Record< StructuredNativeChatBlocker, Exclude > = { + 'reused-terminal': 'reused_terminal', 'agent-without-structured-session': 'agent_without_structured_session', 'draft-prompt': 'structured_unsupported_on_host', 'floating-workspace': 'structured_unsupported_on_host', @@ -89,9 +91,7 @@ const BLOCKER_REASON: Record< 'remote-execution-host': 'remote_execution_host', 'project-runtime': 'wsl_execution_runtime', 'runtime-capability': 'structured_sessions_unavailable', - // Orchestration passes its own host's list, so this is unreachable there; the map is - // exhaustive by type and must still name it. - 'runtime-capability-unknown': 'structured_sessions_unavailable' + 'runtime-capability-unknown': 'structured_support_unknown' } /** The host's own create-support verdict (`agentSession.createSupport`) in this vocabulary. */ @@ -117,15 +117,11 @@ export function decideWorkerStartMode(args: { detail: 'Started a terminal agent worker, the default for new agent tabs in your settings.' } } - const placementReason = resolvePlacementReason(params) - if (placementReason) { - return downgraded(placementReason) - } const agent = params.agent as TuiAgent const support = resolveStructuredNativeChatSupport({ agent, - // Set only by --on, which the placement check above already turned into a fallback. - executionHostId: 'local', + executionHostId: params.on ? `runtime:${params.on}` : 'local', + reusesTerminal: Boolean(params.terminal), hostCapabilities: RUNTIME_CAPABILITIES, // Orchestration resolves a managed worktree or folder workspace; a floating terminal is never // a worker placement. WSL is left to the executing host's own create-support probe, which @@ -169,14 +165,14 @@ async function readStructuredCreateSupport( runtime: Pick, worktreeId: string, agent: TuiAgent | undefined -): Promise<{ supported: boolean; reason?: 'agent' | 'remote' | 'wsl' }> { +): Promise<{ supported: boolean; reason?: 'agent' | 'remote' | 'wsl' } | null> { if (agent !== 'claude' && agent !== 'codex') { return { supported: false, reason: 'agent' } } try { return await runtime.getStructuredAgentSessionCreateSupport(`id:${worktreeId}`, agent) } catch { - return { supported: false } + return null } } @@ -186,34 +182,19 @@ async function readStructuredCreateSupport( */ export function downgradeWorkerStartModeForHost( receipt: WorkerStartModeReceipt, - support: { supported: boolean; reason?: 'agent' | 'remote' | 'wsl' } + support: { supported: boolean; reason?: 'agent' | 'remote' | 'wsl' } | null ): WorkerStartModeReceipt { - if (receipt.mode !== 'structured' || support.supported) { + if (receipt.mode !== 'structured' || support?.supported) { return receipt } + if (support === null) { + return downgraded(BLOCKER_REASON['runtime-capability-unknown']) + } return downgraded( support.reason ? HOST_SUPPORT_REASON[support.reason] : 'structured_unsupported_on_host' ) } -function resolvePlacementReason( - params: WorkerStartModePlacement -): Exclude | null { - if (params.on) { - return 'remote_execution_host' - } - if (params.terminal) { - return 'reused_terminal' - } - if (params.worktree === 'new-child' || params.worktree === 'new-top-level') { - return 'worktree_creation' - } - if (params.model || params.effort) { - return 'launch_preferences' - } - return null -} - function downgraded( reason: Exclude ): WorkerStartModeReceipt { diff --git a/src/main/runtime/rpc/methods/orchestration-worker-support-unknown.test.ts b/src/main/runtime/rpc/methods/orchestration-worker-support-unknown.test.ts new file mode 100644 index 00000000000..1adca072f20 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration-worker-support-unknown.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it, vi } from 'vitest' +import { + decideWorkerStartMode, + resolveWorkerStartModeOnHost +} from './orchestration-worker-start-mode' + +const mode = decideWorkerStartMode({ + params: { agent: 'claude' }, + settings: { + experimentalNativeChat: true, + experimentalStructuredNativeChat: true, + openAgentTabsInChatByDefault: true + } +}) + +describe('host support evidence', () => { + it('distinguishes an unanswered host from an explicit refusal without creating a session', async () => { + const getStructuredAgentSessionCreateSupport = vi + .fn() + .mockRejectedValue(new Error('disconnected')) + const runtime = { getStructuredAgentSessionCreateSupport } + const unknown = await resolveWorkerStartModeOnHost(runtime, mode, 'workspace-1', 'claude') + expect(unknown).toMatchObject({ + mode: 'terminal', + preferred: 'structured', + reason: 'structured_support_unknown' + }) + expect(unknown.detail).toContain('has not established') + expect(getStructuredAgentSessionCreateSupport).toHaveBeenCalledWith('id:workspace-1', 'claude') + getStructuredAgentSessionCreateSupport.mockResolvedValue({ supported: false }) + const refusal = await resolveWorkerStartModeOnHost(runtime, mode, 'workspace-1', 'claude') + expect(refusal.reason).toBe('structured_unsupported_on_host') + expect(refusal.detail).toContain('cannot create') + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration.ts b/src/main/runtime/rpc/methods/orchestration.ts index fbc8f263cd0..ed89ae4519d 100644 --- a/src/main/runtime/rpc/methods/orchestration.ts +++ b/src/main/runtime/rpc/methods/orchestration.ts @@ -1,5 +1,4 @@ import type { RpcMethod } from '../core' -import { sweepingSettledWorkerResumeFences } from './settled-worker-resume-fence-sweep' import { ORCHESTRATION_RUN_METHODS } from './orchestration/runs/runs' import { ORCHESTRATION_WORKER_METHODS } from './orchestration/worker/worker-methods' import { ORCHESTRATION_FEDERATION_METHODS } from './orchestration/federation/federation-methods' @@ -24,4 +23,4 @@ export const ORCHESTRATION_METHODS: RpcMethod[] = [ ...ORCHESTRATION_ASK_METHODS, ...ORCHESTRATION_GATE_METHODS, ...ORCHESTRATION_RESET_METHODS -].map(sweepingSettledWorkerResumeFences) +] diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federated-message-targeting.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federated-message-targeting.test.ts index 8e80a8e3925..c28ef94a4a9 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federated-message-targeting.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federated-message-targeting.test.ts @@ -25,6 +25,7 @@ describe('orchestration federated message targeting', () => { vi.spyOn(runtime, 'getTerminalPaneKey').mockReturnValue(paneKey) vi.spyOn(runtime, 'getTerminalProcessIncarnation').mockReturnValue(processIncarnation) db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: 'task_remote_targeting', homePeerFingerprint: 'home_peer', diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts index f3e160244d1..d5150dc052e 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts @@ -152,6 +152,7 @@ describe('federated worker release ownership', () => { function createAttachment(dispatchId: string, terminalOwnership?: 'created' | 'external'): void { db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: `task_${dispatchId}`, homePeerFingerprint: HOME_FINGERPRINT, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federated-worker-start.ts b/src/main/runtime/rpc/methods/orchestration/federation/federated-worker-start.ts index b577cc87737..a7c59b7064b 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federated-worker-start.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federated-worker-start.ts @@ -162,6 +162,7 @@ export async function startFederatedWorker(args: { server.environmentId, 'orchestration.federationAttachStart', { + runId, dispatchId: started.dispatch.id, taskId: taskForRemote.id, taskSpec: taskForRemote.spec, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-agent-launch.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-agent-launch.test.ts index 748e4c55295..ace3bfe407a 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-agent-launch.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-agent-launch.test.ts @@ -56,6 +56,7 @@ describe('federated worker agent launch', () => { const result = (await method.handler( method.params!.parse({ + runId: 'run-home', dispatchId: 'ctx_remote', taskId: 'task_remote', taskSpec: 'remote cursor worker', diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-control-mail.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-control-mail.test.ts index 755f85fd512..c95ec9b5630 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-control-mail.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-control-mail.test.ts @@ -107,6 +107,7 @@ describe('orchestration federation control mail', () => { homeDb.markWorkerDispatchReady(dispatchId) workerDb.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: task.id, homePeerFingerprint: homeFingerprint, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-effects.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-effects.ts index da0bcf92894..36d25552d80 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-effects.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-effects.ts @@ -29,7 +29,9 @@ export function appendFederationTerminalEffects( : terminal.handle === setupHandle ? 'setup' : 'configured_tab', - action: terminal.handle === agentHandle ? 'reused_agent_terminal' : 'created', + // Remote agent-first worktree creation made every listed terminal, the agent one included; + // the verb is a lifecycle fact, not a role marker. + action: 'created', id: terminal.handle, tabId: terminal.tabId, leafId: terminal.leafId diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-folder-placement.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-folder-placement.test.ts index b8264bd61a7..68364dac1ed 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-folder-placement.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-folder-placement.test.ts @@ -27,6 +27,7 @@ describe('orchestration federated folder placement', () => { await expect( method.handler( method.params!.parse({ + runId: 'run-home', dispatchId: 'ctx_folder', taskId: 'task_folder', taskSpec: 'work in folder', diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts index 3e949383731..e111865d904 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts @@ -445,6 +445,7 @@ describe('orchestration federation lifecycle settlement', () => { const dispatchId = `ctx_persisted_protocol_${protocolVersion}` const taskId = `task_persisted_protocol_${protocolVersion}` workerDb.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId, homePeerFingerprint: 'run-home-device-token', diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts index 20ae3135ec2..7c75c52eb6c 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts @@ -58,6 +58,7 @@ describe('federation host liveness verdicts', () => { status: 'exited' } as never) db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId: DISPATCH_ID, taskId: 'task_remote', homePeerFingerprint: HOME_FINGERPRINT, @@ -112,6 +113,7 @@ describe('federation host liveness verdicts', () => { throw new Error('Expected the real runtime PTY to be listed') } hostDb.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId: DISPATCH_ID, taskId: 'task_remote', homePeerFingerprint: HOME_FINGERPRINT, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-setup.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-setup.test.ts index c905ddffeb8..83865cf96e4 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-setup.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-setup.test.ts @@ -49,6 +49,7 @@ describe('orchestration federated setup evidence', () => { } ] db.createRemoteDispatchAttachment({ + runId: 'run-home', dispatchId, taskId: 'task_remote_setup', homePeerFingerprint: 'home_peer', diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-prompt-budget.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-prompt-budget.test.ts index 83b446eb102..d3d5b6d71b1 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-prompt-budget.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-prompt-budget.test.ts @@ -29,6 +29,7 @@ describe('federation attach-start prompt budget', () => { await expect( method.handler( method.params!.parse({ + runId: 'run-home', dispatchId: 'ctx_oversized_remote', taskId: 'task_oversized_remote', taskSpec: 'x'.repeat(8 * 1024 * 1024), diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.test.ts new file mode 100644 index 00000000000..f00e5102d6d --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, it } from 'vitest' +import { FederationAttachStartParams } from './federation-start-schema' + +// The request shape a v1.4.198 coordinator sends: no runId field at all. +const legacyRequest = { + dispatchId: 'ctx_legacy', + taskId: 'task_legacy', + taskSpec: 'Do the thing', + protocolVersion: 3, + worktree: 'feature-branch' +} + +describe('FederationAttachStartParams', () => { + it('parses a v1.4.198 request that carries no runId', () => { + const result = FederationAttachStartParams.safeParse(legacyRequest) + expect(result.success, result.success ? undefined : JSON.stringify(result.error.issues)).toBe( + true + ) + expect(result.success && result.data.runId).toBeUndefined() + }) + + it('keeps a v1.4.199 runId verbatim', () => { + const result = FederationAttachStartParams.parse({ ...legacyRequest, runId: 'run_home' }) + expect(result.runId).toBe('run_home') + }) + + // Pins OptionalString: '' and non-strings drop to undefined (a stub Run is minted downstream); + // whitespace-only passes the schema and is refused by createRemoteDispatchAttachment. + it('maps an empty or non-string runId to undefined but passes whitespace through', () => { + expect(FederationAttachStartParams.parse({ ...legacyRequest, runId: '' }).runId).toBeUndefined() + expect(FederationAttachStartParams.parse({ ...legacyRequest, runId: 7 }).runId).toBeUndefined() + expect(FederationAttachStartParams.parse({ ...legacyRequest, runId: ' ' }).runId).toBe(' ') + }) + + it('still requires the dispatch, task, spec, and worktree fields', () => { + for (const field of ['dispatchId', 'taskId', 'taskSpec', 'worktree'] as const) { + const { [field]: _dropped, ...rest } = legacyRequest + expect(FederationAttachStartParams.safeParse(rest).success, field).toBe(false) + } + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts index d5d1874a788..84ed57d58cc 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts @@ -3,6 +3,8 @@ import { OptionalFiniteNumber, OptionalString, requiredString } from '../../../s import { OptionalWorkerLaunchPreference } from '../worker/worker-start-schema' export const FederationAttachStartParams = z.object({ + /** Omitted by v1.4.198 coordinators; the worker host then mints a stub home Run. */ + runId: OptionalString, dispatchId: requiredString('Missing Dispatch ID'), taskId: requiredString('Missing Task ID'), taskSpec: requiredString('Missing Task spec'), diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation.ts index 57afd3103a2..785f6a67eec 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation.ts @@ -65,6 +65,7 @@ export const ORCHESTRATION_FEDERATION_ATTACH_METHODS: RpcMethod[] = [ const db = runtime.getOrchestrationDb() db.createRemoteDispatchAttachment({ + runId: params.runId, dispatchId: params.dispatchId, taskId: params.taskId, homePeerFingerprint: orchestrationMutation.callerFingerprint, diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts new file mode 100644 index 00000000000..58e0b3aa0ca --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts @@ -0,0 +1,188 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ORCHESTRATION_METHODS } from '../../orchestration' +import type { RpcContext } from '../../../core' +import { OrchestrationDb } from '../../../../orchestration/db' +import { OrcaRuntimeService } from '../../../../orca-runtime' +import { + encodeFederatedControlMessage, + importFederatedControlMessage +} from '../../../../orchestration/federation-control-message' + +const DISPATCH_ID = 'ctx_federated_worker_1' +const WORKER_HANDLE = 'term_federated_worker' +const WORKER_PANE = 'tab_w:eeeeeeee-eeee-4eee-8eee-eeeeeeeeeeee' +const INCARNATION = 'runtime_test:term_federated_worker:1' + +type CheckResult = { + runId: string + deliveryId: string | null + messages: { id: string; subject: string }[] + count: number + replayed: boolean + acknowledged: string | null +} + +describe('orchestration.check on a federated attachment across a restart', () => { + let directory: string | undefined + let db: OrchestrationDb | undefined + + afterEach(() => { + db?.close() + db = undefined + if (directory) { + rmSync(directory, { recursive: true, force: true }) + directory = undefined + } + }) + + function launch(path: string): RpcContext { + db = new OrchestrationDb(path) + const runtime = new OrcaRuntimeService() + runtime.setOrchestrationDb(db) + vi.spyOn(runtime, 'getTerminalPaneKey').mockImplementation((handle) => + handle === WORKER_HANDLE ? WORKER_PANE : null + ) + vi.spyOn(runtime, 'getLiveTerminalPaneKey').mockImplementation((handle) => + runtime.getTerminalPaneKey(handle) + ) + vi.spyOn(runtime, 'getTerminalProcessIncarnation').mockImplementation((handle) => + handle === WORKER_HANDLE ? INCARNATION : null + ) + return { runtime } + } + + function check(ctx: RpcContext, params: Record = {}): Promise { + const method = ORCHESTRATION_METHODS.find((entry) => entry.name === 'orchestration.check') + if (!method) { + throw new Error('orchestration.check is not registered') + } + const parsed = method.params + ? method.params.parse({ terminal: WORKER_HANDLE, ...params }) + : undefined + return method.handler(parsed, ctx) as Promise + } + + function attach(store: OrchestrationDb, dispatchId: string, runId: string): void { + store.createRemoteDispatchAttachment({ + dispatchId, + runId, + taskId: 'task_federated_1', + homePeerFingerprint: 'peer_fp', + protocolVersion: 1, + runtimeEpoch: 'epoch_1', + mutationReceipt: { + callerFingerprint: 'peer_fp', + requestId: 'attach_1', + method: 'orchestration.federationAttachStart', + payloadHash: 'attach_payload' + } + }) + expect(store.getRunRaw(runId)).toBeDefined() + store.prepareRemoteAttachmentAuthority({ + dispatchId, + paneKey: WORKER_PANE, + processIncarnation: INCARNATION, + worktreeId: 'folder_workspace', + terminalHandle: WORKER_HANDLE, + setupState: 'not_applicable', + effects: [] + }) + store.markRemoteAttachmentReady(dispatchId) + } + + it('replays the coordinator instruction and takes its ack after the app restarts', async () => { + directory = mkdtempSync(join(tmpdir(), 'orca-federated-check-')) + const path = join(directory, 'orchestration.db') + + const first = launch(path) + attach(db as OrchestrationDb, DISPATCH_ID, 'run_coordinator') + importFederatedControlMessage(db as OrchestrationDb, { + dispatchId: DISPATCH_ID, + messageId: 'msg_federated_1', + payload: encodeFederatedControlMessage({ + from: 'term_coord', + subject: 'continue the task', + body: 'the plan changed', + type: 'dispatch', + priority: 'normal', + threadId: null, + payload: null + }) + }) + + const delivered = await check(first) + expect(delivered.messages.map((message) => message.id)).toEqual(['msg_federated_1']) + expect(delivered.runId).toBe('run_coordinator') + expect(delivered.replayed).toBe(false) + const deliveryId = delivered.deliveryId as string + expect(deliveryId).not.toBeNull() + ;(db as OrchestrationDb).close() + + // The worker's process outlives the app; its instruction is still unacknowledged. + const second = launch(path) + const replayed = await check(second) + expect(replayed.deliveryId).toBe(deliveryId) + expect(replayed.replayed).toBe(true) + expect(replayed.messages.map((message) => message.id)).toEqual(['msg_federated_1']) + + const acknowledged = await check(second, { ack: deliveryId }) + expect(acknowledged.acknowledged).toBe(deliveryId) + expect(acknowledged.count).toBe(0) + }) + + it('files loopback mail once under the local Dispatch Run without replacing its owner', async () => { + const ctx = launch(':memory:') + const store = db as OrchestrationDb + const run = store.createRun({ + objective: 'loopback coordinator', + coordinatorHandle: 'term_coord', + coordinatorPaneKey: 'tab_coord:pane_coord' + }) + const task = store.createTask({ runId: run.id, spec: 'loopback task' }) + const { dispatch } = store.createStartingWorkerDispatch({ + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + taskId: task.id, + startOptions: {} + }) + attach(store, dispatch.id, run.id) + expect(store.getRemoteDispatchAttachment(dispatch.id)?.home_run_id).toBe(dispatch.run_id) + expect(store.getRun(run.id)).toEqual(run) + const message = { + dispatchId: dispatch.id, + messageId: 'msg_loopback', + payload: encodeFederatedControlMessage({ + from: 'term_coord', + subject: 'continue', + body: 'loopback instruction', + type: 'dispatch', + priority: 'normal', + threadId: null, + payload: null + }) + } + expect(importFederatedControlMessage(store, message).imported).toBe(true) + expect(importFederatedControlMessage(store, message).imported).toBe(false) + expect(store.getMessageById(message.messageId)?.run_id).toBe(run.id) + const delivered = await check(ctx) + expect(delivered.runId).toBe(run.id) + expect(delivered.messages.map((entry) => entry.id)).toEqual([message.messageId]) + expect((await check(ctx, { ack: delivered.deliveryId })).count).toBe(0) + }) + + it('refuses an attachment with no home Run before writing a Delivery', async () => { + const ctx = launch(':memory:') + const store = db as OrchestrationDb + attach(store, DISPATCH_ID, 'run_coordinator') + const attachment = store.getRemoteDispatchAttachment(DISPATCH_ID)! + vi.spyOn(store, 'findActiveRemoteAttachmentForPane').mockReturnValue({ + ...attachment, + home_run_id: undefined + } as never) + await expect(check(ctx)).rejects.toThrow() + expect(store.db.prepare('SELECT id FROM deliveries').all()).toEqual([]) + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker.ts b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker.ts index 27df8fd2afa..355a12be5c1 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker.ts @@ -3,7 +3,6 @@ import type { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { formatMessageBanner } from '../../../../orchestration/formatter' import { exposeMessages } from './mailbox-message-receipt' -import { ORCHESTRATION_LEGACY_RUN_ID } from '../../../../../../shared/orchestration-rpc-contract' import { routeAllMailboxPages } from '../schemas' import { asDispatchFence, callerHoldsDispatchPane, dispatchFenced } from './dispatch-mailbox-fence' import type { CheckParams } from '../schemas' @@ -46,13 +45,15 @@ export async function checkWorkerMailbox(args: { : remoteAttachment ? { dispatchId: remoteAttachment.dispatch_id, - runId: undefined, + runId: remoteAttachment.home_run_id, generation: remoteAttachment.consumer_generation } : undefined if (!workerMailbox) { return undefined } + const deliveryRunId = workerMailbox.runId + db.requireRun(deliveryRunId) const address = `dispatch:${workerMailbox.dispatchId}` // Why: a federated worker host has no dispatch_contexts row, so its generation lives on the // remote_dispatch_attachments row instead. @@ -164,7 +165,6 @@ export async function checkWorkerMailbox(args: { } } await revalidateWorkerMailbox() - const deliveryRunId = workerMailbox.runId ?? ORCHESTRATION_LEGACY_RUN_ID let acknowledged try { acknowledged = params.ack diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-point-to-point.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-point-to-point.ts index c7386acd49f..807e709003a 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/send-point-to-point.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-point-to-point.ts @@ -7,7 +7,6 @@ import type { SendParams } from '../schemas' import { legacyWorkerDeliveryContract } from '../routing' import { exposeMessage } from './mailbox-message-receipt' import { recordReceiptForPostCommitNudge } from './mutation-replay-nudge' -import { sweepSettledWorkerResumeFences } from '../../settled-worker-resume-fence-sweep' import type { SendRecipientWarning } from './recipient-routing' import type { z } from 'zod' @@ -150,11 +149,6 @@ export function sendPointToPointMessage(args: { ? db.commitWorkerDoneMessageMutation(commitMessage) : commitMessage() committed.nudge() - if (messageType === 'worker_done') { - // Settlement is what makes the pane fenceable; without this the fence only appeared at the - // next app start and reopening the pane in the same session respawned the agent. - sweepSettledWorkerResumeFences(runtime) - } return committed.receipt } diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-unbound-terminals.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-unbound-terminals.test.ts new file mode 100644 index 00000000000..84cb250a19e --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-unbound-terminals.test.ts @@ -0,0 +1,31 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { createOrchestrationRpcHarness } from '../rpc-test-harness' + +describe('orchestration.send between terminals in no Run', () => { + const h = createOrchestrationRpcHarness() + afterEach(() => h.cleanup()) + + it('delivers terminal-to-terminal mail when neither terminal is in a Run', async () => { + // Two plain panes and `send --to `: the first command the guide teaches. #19542 + // refused this with a bare "Run is required"; it files under the unbound Run instead. + const { db, runtime, ctx } = h.setup(false) + vi.spyOn(runtime, 'getTerminalPaneKey').mockImplementation((handle) => + handle === 'term_a' ? 'tab_a:leaf_a' : handle === 'term_b' ? 'tab_b:leaf_b' : null + ) + vi.spyOn(runtime, 'deliverPendingMessagesForHandle').mockImplementation(() => {}) + + const result = (await h.call( + 'orchestration.send', + { from: 'term_a', to: 'term_b', subject: 'hello from no Run' }, + ctx + )) as { message: { id: string; run_id: string; to_handle: string } } + + expect(result.message).toMatchObject({ run_id: 'run_unbound', to_handle: 'term_b' }) + expect(db.getRun('run_unbound')).toMatchObject({ legacy: 0 }) + expect(db.getUnreadMessages('term_b').map((row) => row.id)).toEqual([result.message.id]) + const checked = (await h.call('orchestration.check', { terminal: 'term_b' }, ctx)) as { + messages: { id: string }[] + } + expect(checked.messages.map((row) => row.id)).toEqual([result.message.id]) + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/runs/migration-behavior.test.ts b/src/main/runtime/rpc/methods/orchestration/runs/migration-behavior.test.ts index d372c733246..929dd5c1ee3 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/migration-behavior.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/migration-behavior.test.ts @@ -31,7 +31,7 @@ describe('orchestration migration behavior', () => { it('lists an explicitly selected legacy Run without binding or mutation', async () => { const { db, runtime } = createRuntime() - const task = db.createTask({ spec: 'pre-upgrade work' }) + const task = db.createTask({ runId: 'run_legacy_local', spec: 'pre-upgrade work' }) const taskList = ORCHESTRATION_METHODS.find( (method) => method.name === 'orchestration.taskList' )! @@ -55,6 +55,7 @@ describe('orchestration migration behavior', () => { it('formats legacy terminal inspection as read-only without consuming mail', async () => { const { db, runtime } = createRuntime() const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coord', subject: 'still working', @@ -79,6 +80,7 @@ describe('orchestration migration behavior', () => { // A consuming check refuses a handle with no live pane before it reads any mail. vi.spyOn(runtime, 'getTerminalPaneKey').mockReturnValue('tab_legacy:leaf_legacy') const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coord', subject: 'still working' @@ -98,6 +100,7 @@ describe('orchestration migration behavior', () => { it('rejects replies to legacy mail without marking or inserting rows', async () => { const { db, runtime } = createRuntime() const message = db.insertMessage({ + runId: 'run_legacy_local', from: 'term_worker', to: 'term_coord', subject: 'legacy question' diff --git a/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts index 32863c09371..aee45e25259 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts @@ -572,7 +572,7 @@ describe('orchestration RPC methods', () => { ) expect(result.effects).toEqual( expect.arrayContaining([ - expect.objectContaining({ role: 'agent', action: 'reused_agent_terminal' }), + expect.objectContaining({ role: 'agent', action: 'created' }), expect.objectContaining({ role: 'setup', action: 'created' }), expect.objectContaining({ role: 'configured_tab', action: 'created' }) ]) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/created-worker-terminal-custody.ts b/src/main/runtime/rpc/methods/orchestration/worker/created-worker-terminal-custody.ts new file mode 100644 index 00000000000..c73884b1d96 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/created-worker-terminal-custody.ts @@ -0,0 +1,32 @@ +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import type { OrchestrationDb } from '../../../../orchestration/db' +import { requireWorkerAuthority } from './worker-topology' + +/** + * Custody for an agent terminal this start created, recorded when the terminal exists rather than + * after the agent boot wait: a keystroke into the booting pane has to find an `owned` row to flip, + * or the takeover is dropped and a later `worker-release` closes the pane under the user. + * + * Ownership of a pane only. The Dispatch capability still waits for the agent to come up. + * + * `created` is false for an explicit `--terminal` reuse, which is the caller's own pane, and for a + * structured session, which reaches its authority in this same turn and so has no gap to close. + */ +export function recordCreatedWorkerTerminalCustody( + runtime: OrcaRuntimeService, + stage: { db: OrchestrationDb; dispatchId: string; worktreeId: string; terminalHandle: string }, + created: boolean +): void { + if (!created) { + return + } + const authority = requireWorkerAuthority(runtime, stage.terminalHandle) + stage.db.recordCreatedWorkerTerminalCustody({ + dispatchId: stage.dispatchId, + handle: stage.terminalHandle, + paneKey: authority.paneKey, + processIncarnation: authority.processIncarnation, + worktreeId: stage.worktreeId, + hostScope: authority.hostScope ?? null + }) +} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.test.ts deleted file mode 100644 index bdf5daad565..00000000000 --- a/src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.test.ts +++ /dev/null @@ -1,191 +0,0 @@ -import { afterEach, describe, expect, it } from 'vitest' -import type { OrcaRuntimeService } from '../../../../orca-runtime' -import { OrchestrationDb } from '../../../../orchestration/db' -import { resolveResidualAgentTerminal } from './failed-start-residual-terminal' -import { failWorkerStartWithReceipt } from './worker-start-receipt' -import type { WorkerEffect } from './worker-topology' - -const HANDLE = 'term_residual' -const PANE_KEY = 'tab_residual:leaf_residual' -const INCARNATION = 'pty-residual:1' - -const createdAgentTerminal: WorkerEffect = { - kind: 'terminal', - role: 'agent', - action: 'created', - id: HANDLE, - surface: 'visible' -} - -function createRuntime(overrides: Partial> = {}): OrcaRuntimeService { - return { - getOrchestrationDispatchAuthority: () => ({ - paneKey: PANE_KEY, - processIncarnation: INCARNATION, - hostScope: { kind: 'local', hostId: 'local' } - }), - getTerminalPaneKey: () => PANE_KEY, - getTerminalProcessIncarnation: () => INCARNATION, - ...overrides - } as unknown as OrcaRuntimeService -} - -describe('residual agent terminal left by a failed start', () => { - it('resolves identity for a terminal this start created', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime(), - effects: [createdAgentTerminal], - terminalHandle: HANDLE, - worktreeId: 'repo::worktree' - }) - ).toEqual({ - terminalHandle: HANDLE, - worktreeId: 'repo::worktree', - paneKey: PANE_KEY, - processIncarnation: INCARNATION, - hostScope: JSON.stringify({ kind: 'local', hostId: 'local' }) - }) - }) - - it('resolves the agent-first worktree terminal the same way', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime(), - effects: [{ ...createdAgentTerminal, action: 'reused_agent_terminal' }], - terminalHandle: HANDLE, - worktreeId: null - }) - ).toMatchObject({ terminalHandle: HANDLE }) - }) - - it('never claims a caller-supplied terminal', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime(), - effects: [{ ...createdAgentTerminal, action: 'reused' }], - terminalHandle: HANDLE, - worktreeId: null - }) - ).toBeUndefined() - }) - - it('never claims a setup terminal', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime(), - effects: [{ ...createdAgentTerminal, role: 'setup' }], - terminalHandle: HANDLE, - worktreeId: null - }) - ).toBeUndefined() - }) - - it('refuses a pane whose process cannot be identified', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime({ - getOrchestrationDispatchAuthority: () => null, - getTerminalProcessIncarnation: () => null - }), - effects: [createdAgentTerminal], - terminalHandle: HANDLE, - worktreeId: null - }) - ).toBeUndefined() - }) - - it('refuses when the start never resolved a terminal', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime(), - effects: [], - terminalHandle: undefined, - worktreeId: null - }) - ).toBeUndefined() - }) - - it('stays silent when identity resolution throws', () => { - expect( - resolveResidualAgentTerminal({ - runtime: createRuntime({ - getOrchestrationDispatchAuthority: () => { - throw new Error('handle retired') - } - }), - effects: [createdAgentTerminal], - terminalHandle: HANDLE, - worktreeId: null - }) - ).toBeUndefined() - }) -}) - -describe('failed worker-start receipt for a residual terminal', () => { - let db: OrchestrationDb | undefined - - afterEach(() => { - db?.close() - }) - - function failStart(residual: boolean): { recovery?: string } { - const d = (db = new OrchestrationDb(':memory:')) - const task = d.createTask({ spec: 'residual receipt' }) - const started = d.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: task.id, - startOptions: {} - }) - d.recordWorkerStage({ - dispatchId: started.dispatch.id, - stage: 'terminal_readying', - terminalHandle: HANDLE, - effects: [createdAgentTerminal], - residualResources: [createdAgentTerminal] - }) - return failWorkerStartWithReceipt({ - db: d, - mode: { - mode: 'terminal', - preferred: 'terminal', - reason: 'user_default', - detail: 'terminal by default' - } as const, - runId: 'run_residual', - taskId: task.id, - dispatchId: started.dispatch.id, - failedStage: 'agent_readiness', - error: new Error('Agent startup blocked: codex-interactive-prompt'), - setup: { - requested: 'not_applicable', - effective: 'not_applicable', - source: 'existing_worktree', - hookFound: false, - startupPolicy: 'start-immediately', - state: 'not_applicable' - }, - launch: { requested: { agent: 'codex' }, effective: { agent: 'codex' } } as never, - ...(residual - ? { - residualAgentTerminal: { - terminalHandle: HANDLE, - worktreeId: 'repo::worktree', - paneKey: PANE_KEY, - processIncarnation: INCARNATION, - hostScope: null - } - } - : {}) - }) as { recovery?: string } - } - - it('names worker-release for the terminal it left behind', () => { - expect(failStart(true).recovery).toContain('worker-release') - }) - - it('promises no cleanup when there is no residual terminal', () => { - expect(failStart(false).recovery).toBeUndefined() - }) -}) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.ts b/src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.ts deleted file mode 100644 index e42923e93a9..00000000000 --- a/src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.ts +++ /dev/null @@ -1,53 +0,0 @@ -import type { OrcaRuntimeService } from '../../../../orca-runtime' -import type { FailedStartTerminalAdoption } from '../../../../orchestration/db/worker-terminal/failed-start-terminal-adoption' -import type { WorkerEffect } from './worker-topology' - -/** True only for an agent terminal this worker-start brought into existence. An explicit - * `--terminal` reuse records `reused` and is never residual — it is the caller's terminal. */ -function orchestrationCreatedAgentTerminal( - effects: readonly WorkerEffect[], - handle: string -): boolean { - return effects.some( - (effect) => - effect.kind === 'terminal' && - effect.role === 'agent' && - effect.id === handle && - (effect.action?.startsWith('created') === true || effect.action === 'reused_agent_terminal') - ) -} - -/** - * Identity for the terminal a failed start leaves behind, so the failed Dispatch can own it and - * `worker-release` can close it. Returns nothing unless the pane and process are both provable: - * an unprovable identity must never authorize a later close. - */ -export function resolveResidualAgentTerminal(args: { - runtime: OrcaRuntimeService - effects: readonly WorkerEffect[] - terminalHandle: string | undefined - worktreeId: string | null -}): FailedStartTerminalAdoption | undefined { - const handle = args.terminalHandle - if (!handle || !orchestrationCreatedAgentTerminal(args.effects, handle)) { - return undefined - } - try { - const authority = args.runtime.getOrchestrationDispatchAuthority(handle) - const paneKey = authority?.paneKey ?? args.runtime.getTerminalPaneKey(handle) - const processIncarnation = - authority?.processIncarnation ?? args.runtime.getTerminalProcessIncarnation(handle) - if (!paneKey || !processIncarnation) { - return undefined - } - return { - terminalHandle: handle, - worktreeId: args.worktreeId, - paneKey, - processIncarnation, - hostScope: authority?.hostScope ? JSON.stringify(authority.hostScope) : null - } - } catch { - return undefined - } -} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/failed-worker-start-teardown.ts b/src/main/runtime/rpc/methods/orchestration/worker/failed-worker-start-teardown.ts index 32827377b54..250b639fc60 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/failed-worker-start-teardown.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/failed-worker-start-teardown.ts @@ -3,40 +3,27 @@ import { discardStructuredWorkerSession, releaseStructuredWorkerSession } from '../../orchestration-structured-worker-session' -import { resolveResidualAgentTerminal } from './failed-start-residual-terminal' import type { createStructuredWorkerSessionForWorktree } from './worker-topology' -import type { FailedStartTerminalAdoption } from '../../../../orchestration/db/worker-terminal/failed-start-terminal-adoption' /** - * Undoes what a start created before it failed, and reports what `worker-release` still owns. + * Undoes what a start created before it failed. * * A start that never reached ready leaves no settlement to release the hold later, and its session * was already published as a chat tab — without the discard, a failed start strands a dead chat tab * that the durable restore index republishes on every app launch. Both halves are best-effort by * construction, so neither can replace the real error. + * + * A created PTY terminal is deliberately NOT torn down: its custody row was written at creation, so + * `worker-release` on the failed Dispatch owns that cleanup and the coordinator decides when. */ export async function tearDownFailedWorkerStart(args: { runtime: OrcaRuntimeService structuredSession: Awaited> | null dispatchId: string - effects: unknown[] - terminalHandle: string | undefined - worktreeId: string | null -}): Promise { +}): Promise { const { runtime, structuredSession } = args - // A structured session is torn down outright here, so it must never also be adopted as a residual - // terminal for `worker-release` to close a second time. - const residualAgentTerminal = structuredSession - ? undefined - : resolveResidualAgentTerminal({ - runtime, - effects: args.effects as never, - terminalHandle: args.terminalHandle, - worktreeId: args.worktreeId - }) releaseStructuredWorkerSession(args.dispatchId, runtime) if (structuredSession) { await discardStructuredWorkerSession(structuredSession.identity.sessionId, runtime) } - return residualAgentTerminal } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/legacy-dispatch-projection.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/legacy-dispatch-projection.test.ts index 4df73994beb..75ad57f3a12 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/legacy-dispatch-projection.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/legacy-dispatch-projection.test.ts @@ -117,6 +117,6 @@ describe('pre-v3 dispatch rows in worker-list', () => { }) expect(worker.projection.attention.categories).toContain('unverifiable') expect(worker.projection.attention.requiresAction).toBe(true) - expect(worker.projection.nextAction.kind).toBe('inspect') + expect(worker.projection.nextAction).toEqual({ kind: 'none', argv: [] }) }) }) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts b/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts index 48b14f9a84e..ec188695f5c 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts @@ -1,4 +1,3 @@ -import type { TuiAgent } from '../../../../../../shared/tui-agent' import type { OrcaRuntimeService } from '../../../../orca-runtime' import type { OrchestrationDb } from '../../../../orchestration/db' import type { RunRow, TaskRow } from '../../../../orchestration/types' @@ -8,6 +7,8 @@ import { resolveWorkerStartModeOnHost, type WorkerStartModeReceipt } from '../../orchestration-worker-start-mode' +import { EXISTING_WORKTREE_SETUP, placeWorkerAgent } from './worker-start-agent-placement' +import { awaitStructuredWorkerSetupGate } from './worker-start-structured-setup-gate' import { assertOrchestrationWorktreeCreationSupported } from './folder-worktree-placement' import type { WorkerStartInput } from './worker-start-schema' import { @@ -18,18 +19,11 @@ import { import { failWorkerStartWithReceipt } from './worker-start-receipt' import { parseTaskDeps } from './task-deps-argument' import { assertExplicitWorkerTerminalUsable } from './explicit-worker-terminal-validation' -import { deliverWorkerDispatchPreamble } from './deliver-worker-dispatch-preamble' +import { recordCreatedWorkerTerminalCustody } from './created-worker-terminal-custody' import { tearDownFailedWorkerStart } from './failed-worker-start-teardown' -import { - createExistingWorktreeWorkerTerminal, - createStructuredWorkerSessionForWorktree, - createWorkerWorktree, - monitorWorkerSetup, - requireWorkerAuthority, - type WorkerEffect, - type WorkerSetupReceipt -} from './worker-topology' +import { requireWorkerAuthority, type WorkerEffect } from './worker-topology' import { prepareLocalWorkerStart } from './worker-start-validation' +import { deliverAndSettleWorkerStartReadiness } from './worker-start-readiness-settlement' type WorkerStartMutation = { callerFingerprint: string @@ -79,7 +73,7 @@ export async function startLocalWorker(args: { resolvedWorktreeId: resolvedWorktree?.id }) } - const mode = await resolveWorkerStartModeOnHost(runtime, args.mode, resolvedWorktree?.id, agent) + let mode = await resolveWorkerStartModeOnHost(runtime, args.mode, resolvedWorktree?.id, agent) const startOptions = { worktree: requestedWorktree, @@ -127,74 +121,33 @@ export async function startLocalWorker(args: { ) } let terminalHandle = params.terminal - let structuredSession: Awaited< - ReturnType - > | null = null - let terminalRevealWarning: string | undefined + let placed: Awaited> | undefined let failedStage = 'terminal_create' - let setupReceipt: WorkerSetupReceipt = { - requested: 'not_applicable', - effective: 'not_applicable', - source: 'existing_worktree', - hookFound: false, - startupPolicy: 'start-immediately', - state: 'not_applicable' - } try { - if (creationWorktree) { - failedStage = 'worktree_create' - const created = await createWorkerWorktree({ - runtime, - db, - dispatchId: started.dispatch.id, - requestedWorktree, - coordinatorWorktree: creationWorktree, - params, - agent: agent as TuiAgent, - launchPreferences: launch.preferences, - effects - }) - resolvedWorktree = created.worktree - terminalHandle = created.terminalHandle - setupReceipt = created.setupReceipt - } else if (!terminalHandle && mode.mode === 'structured') { - db.recordWorkerStage({ - dispatchId: started.dispatch.id, - stage: 'terminal_creating', - worktreeId: resolvedWorktree!.id, - effects - }) - structuredSession = await createStructuredWorkerSessionForWorktree({ - runtime, - worktreeId: resolvedWorktree!.id, - agent: agent as TuiAgent, - dispatchId: started.dispatch.id, - effects - }) - terminalHandle = structuredSession.identity.handle - } else if (!terminalHandle) { - db.recordWorkerStage({ - dispatchId: started.dispatch.id, - stage: 'terminal_creating', - worktreeId: resolvedWorktree!.id, - effects - }) - const terminal = await createExistingWorktreeWorkerTerminal({ - runtime, - worktreeId: resolvedWorktree!.id, - agent: agent as TuiAgent, - launchPreferences: launch.preferences, - taskId: task.id, - effects - }) - terminalHandle = terminal.handle - terminalRevealWarning = terminal.warning - } else { - effects.push({ kind: 'terminal', role: 'agent', action: 'reused', id: terminalHandle }) - } - if (!resolvedWorktree || !terminalHandle) { - throw new Error('Worker topology did not resolve an agent terminal and worktree.') - } + placed = await placeWorkerAgent({ + runtime, + db, + dispatchId: started.dispatch.id, + taskId: task.id, + params, + requestedWorktree, + creationWorktree, + resolvedWorktree, + mode, + agent, + launchPreferences: launch.preferences, + effects, + onStage: (stage) => { + failedStage = stage + } + }) + // A created worktree settles its mode only once the host can be asked about it, so the + // receipt the caller decided is not always the one that ran. + mode = placed.mode + resolvedWorktree = placed.worktree + terminalHandle = placed.terminalHandle + const structuredSession = placed.structuredSession + const setupReceipt = placed.setupReceipt const setupStage = { db, dispatchId: started.dispatch.id, @@ -203,6 +156,7 @@ export async function startLocalWorker(args: { setup: setupReceipt, effects } + recordCreatedWorkerTerminalCustody(runtime, setupStage, !params.terminal && !structuredSession) if (persistGatedSetupSpawnFailure(setupStage)) { failedStage = 'setup_start' throw new Error('Setup terminal failed to start before the gated agent launch.') @@ -211,12 +165,20 @@ export async function startLocalWorker(args: { failedStage = 'agent_readiness' // A structured session is ready the moment its attach returns ok: there is no boot-to-idle - // gap and no terminal title to read an idle edge from. - if (!structuredSession) { - const wait = await runtime.waitForTerminal(terminalHandle, { - condition: 'tui-idle', - timeoutMs: params.timeoutMs ?? 60_000 - }) + // gap and no terminal title to read an idle edge from. Only the repo's wait-for-setup policy + // still holds it back, and that gate has to be waited on explicitly here. + const wait = structuredSession + ? await awaitStructuredWorkerSetupGate({ + runtime, + setup: setupReceipt, + effects, + timeoutMs: params.timeoutMs ?? 60_000 + }) + : await runtime.waitForTerminal(terminalHandle, { + condition: 'tui-idle', + timeoutMs: params.timeoutMs ?? 60_000 + }) + if (wait) { persistWorkerSetupWaitOutcome({ ...setupStage, wait }) if (!wait.satisfied) { if (setupReceipt.state === 'failed') { @@ -225,7 +187,9 @@ export async function startLocalWorker(args: { throw new Error( wait.blockedReason ? `Agent startup blocked: ${wait.blockedReason}` - : `Agent did not become ready (${wait.status}).` + : structuredSession + ? `Setup did not finish before the structured worker started (${wait.status}).` + : `Agent did not become ready (${wait.status}).` ) } } @@ -240,58 +204,35 @@ export async function startLocalWorker(args: { terminalOwnership: params.terminal ? 'external' : 'created' }) - failedStage = 'dispatch_input' - const promptDelivery = await deliverWorkerDispatchPreamble({ + return await deliverAndSettleWorkerStartReadiness({ runtime, - structuredSession, - terminalHandle, + db, + run, + task, dispatchId: started.dispatch.id, dispatchDepth: started.dispatch.depth, - taskId: task.id, - taskSpec: task.spec, + structuredSession, + terminalHandle, coordinatorHandle: params.from, dispatchCapability: capability, devMode: params.devMode, - requestId: orchestrationMutation?.requestId ?? started.dispatch.id - }) - effects.push({ - kind: 'dispatch_input', - role: 'agent', - id: terminalHandle, - state: 'accepted' - }) - const worker = db.markWorkerDispatchReady(started.dispatch.id, effects) - monitorWorkerSetup({ - runtime, - db, - runId: run.id, - dispatchId: started.dispatch.id, + requestId: orchestrationMutation?.requestId ?? started.dispatch.id, + agent: agent ?? null, setupReceipt, - effects - }) - return { - runId: run.id, - taskId: task.id, - dispatchId: started.dispatch.id, - state: worker.state, - stage: worker.stage, - setup: setupReceipt, - launch: launch.receipt, + launchReceipt: launch.receipt, mode, timeoutMs: params.timeoutMs ?? 60_000, effects, - ...(promptDelivery ? { prompt: promptDelivery } : {}), - residualResources: [], - ...(terminalRevealWarning ? { warning: terminalRevealWarning } : {}) - } + terminalRevealWarning: placed.warning, + onStage: (stage) => { + failedStage = stage + } + }) } catch (error) { - const residualAgentTerminal = await tearDownFailedWorkerStart({ + await tearDownFailedWorkerStart({ runtime, - structuredSession, - dispatchId: started.dispatch.id, - effects, - terminalHandle, - worktreeId: resolvedWorktree?.id ?? null + structuredSession: placed?.structuredSession ?? null, + dispatchId: started.dispatch.id }) return failWorkerStartWithReceipt({ db, @@ -300,10 +241,9 @@ export async function startLocalWorker(args: { dispatchId: started.dispatch.id, failedStage, error, - setup: setupReceipt, + setup: placed?.setupReceipt ?? EXISTING_WORKTREE_SETUP, launch: launch.receipt, - mode, - ...(residualAgentTerminal ? { residualAgentTerminal } : {}) + mode }) } } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts index 4cd8810ad6b..2890fa08938 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts @@ -242,7 +242,13 @@ describe('manual Dispatch observation', () => { const result = (await workerListMethod.handler( workerListMethod.params?.parse({ run: run.id }), { runtime } - )) as { workers: { dispatchId: string; workerState: string; terminalState: string | null }[] } + )) as { + workers: { + dispatchId: string + workerState: string + terminalState: string | null + }[] + } expect(result.workers).toEqual([ expect.objectContaining({ @@ -262,7 +268,10 @@ describe('manual Dispatch observation', () => { const runtime = new OrcaRuntimeService() runtime.setOrchestrationDb(db) const closeTerminal = vi.spyOn(runtime, 'closeTerminal') - const task = db.createTask({ spec: 'operator-owned lane' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'operator-owned lane' + }) const dispatch = createRootDispatch( db, task.id, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/structured-worker-launch-seed-options.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/structured-worker-launch-seed-options.test.ts new file mode 100644 index 00000000000..ccad6f3be5f --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/structured-worker-launch-seed-options.test.ts @@ -0,0 +1,104 @@ +/** + * `--model`/`--effort` used to downgrade a structured-preferring worker to a PTY terminal because + * "launch preferences apply only to a terminal agent". They no longer do: the same two ids a saved + * selection seeds a chat with are seeded into the worker's own session here. + */ + +import { describe, expect, it, vi } from 'vitest' + +const createStructuredWorkerSession = vi.fn(async (_args: Record) => ({ + identity: { handle: 'structworker_1', sessionId: 'sess_1' }, + host: {} +})) + +vi.mock('../../orchestration-structured-worker-session', () => ({ + createStructuredWorkerSession: (args: never) => createStructuredWorkerSession(args) +})) + +const { createStructuredWorkerSessionForWorktree } = await import('./worker-topology') +const { prepareStructuredAgentSessionCreateForWorktree } = + await import('../../structured-agent-session-create') + +async function createWith(launchPreferences?: Record) { + createStructuredWorkerSession.mockClear() + await createStructuredWorkerSessionForWorktree({ + runtime: {} as never, + worktreeId: 'repo::wt', + agent: 'codex', + dispatchId: 'ctx_1', + ...(launchPreferences ? { launchPreferences } : {}), + effects: [] + }) + return createStructuredWorkerSession.mock.calls[0]?.[0] ?? {} +} + +describe('a structured worker seeds the dispatch launch preferences', () => { + it('carries --model and --effort into the session create', async () => { + expect(await createWith({ model: 'gpt-5.6-sol', effort: 'high' })).toMatchObject({ + options: { model: 'gpt-5.6-sol', effort: 'high' } + }) + }) + + it('carries only the model when no --effort was asked for', async () => { + expect((await createWith({ model: 'gpt-5.6-sol' })).options).toEqual({ model: 'gpt-5.6-sol' }) + }) + + it.each([ + ['no preferences at all', undefined], + ['an option set that narrows to nothing', { model: ' ' }] + ])('omits options entirely for %s, never sending {}', async (_name, preferences) => { + // `{}` fails the durable record's bounded-string guard, and `agent_session_options_invalid` is + // not a wire refusal code — the throw strands the launch with no fallback. Omitted, the host + // seeds the user's own saved selection instead, which is what a chat would get. + expect(await createWith(preferences)).not.toHaveProperty('options') + }) +}) + +describe('the create the seed options land in', () => { + const settingsResolved = { + location: { executionHostId: 'local', wslDistro: null, workspaceId: 'repo::wt' }, + provider: 'codex', + agent: 'codex', + accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + runtimeKind: 'native', + options: { model: 'saved-model', effort: 'low' } + } + + async function prepare(options?: Record) { + const prepared = await prepareStructuredAgentSessionCreateForWorktree({ + runtime: { + resolveStructuredAgentSessionCreateIntent: async () => settingsResolved + } as never, + ensureHost: async () => ({}) as never, + envelope: { + sessionId: 'sess_1', + clientOperationId: 'op_1', + expectedRuntimeFence: null, + payloadFingerprint: '' + }, + worktree: 'id:repo::wt', + agent: 'codex', + caller: { callerKey: 'orchestration:dispatch:ctx_1' }, + ...(options ? { options } : {}) + }) + return prepared.attachParams + } + + it("replaces the saved selection the host resolved with the dispatch's own", async () => { + expect((await prepare({ model: 'gpt-5.6-sol', effort: 'high' })).options).toEqual({ + model: 'gpt-5.6-sol', + effort: 'high' + }) + }) + + it('keeps the saved selection when the dispatch named none', async () => { + expect((await prepare()).options).toEqual({ model: 'saved-model', effort: 'low' }) + }) + + it('does not let the seed options move the attach fingerprint', async () => { + // Options are the session's initial state, not its identity: a retry re-resolves them and must + // replay rather than conflict. + const [seeded, unseeded] = await Promise.all([prepare({ model: 'gpt-5.6-sol' }), prepare()]) + expect(seeded.envelope.payloadFingerprint).toBe(unseeded.envelope.payloadFingerprint) + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts new file mode 100644 index 00000000000..68d40b4a04e --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts @@ -0,0 +1,165 @@ +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { createOrchestrationWorkerReleaseHarness } from './worker-release.test-support' +import { TERMINAL_SEND_METHODS } from '../../terminal/terminal-send-method' +import { sendTerminalStreamInput } from '../../terminal/terminal-input-delivery' +import { isStreamingMethod, type RpcMethod } from '../../../core' + +const h = createOrchestrationWorkerReleaseHarness() +beforeEach(() => h.setup()) +afterEach(() => h.cleanup()) + +it.each(['local', 'ssh'])( + 'a handle-addressed phone report fences %s worker release', + async (host) => { + if (host === 'ssh') { + vi.mocked(h.runtime.getOrchestrationDispatchAuthority).mockImplementation((handle) => + handle === 'term_worker' + ? ({ + terminalHandle: handle, + paneKey: h.workerPaneKey, + processIncarnation: 'runtime_test:term_worker:1', + hostScope: { kind: 'ssh', targetId: 'ssh-1' } + } as never) + : null + ) + } + const worker = await h.startSettledWorker() + expect(h.db.getWorkerTerminalResourceByOwner(worker.dispatchId)?.host_scope).toContain(host) + h.runtime.registerPreAllocatedHandleForPty('pty-worker', 'term_worker') + h.runtime.registerPty('pty-worker', 'repo::worktree', undefined, { + tabId: 'tab_worker', + leafId: 'bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' + }) + vi.mocked(h.runtime.getTerminalPaneKey).mockRestore() + await expect( + h.call('orchestration.workerTerminalUserInput', { terminal: 'term_worker' }) + ).resolves.toEqual({ changed: 1 }) + expect(h.db.getWorkerTerminalResourceByOwner(worker.dispatchId)?.ownership_state).toBe( + 'user_owned' + ) + await expect( + h.call('orchestration.workerTerminalUserInput', { terminal: 'term_worker' }) + ).resolves.toEqual({ changed: 0 }) + await expect( + h.call('orchestration.workerRelease', { dispatch: worker.dispatchId }) + ).resolves.toMatchObject({ state: 'retained', reason: 'user_takeover' }) + expect(h.runtime.closeTerminal).not.toHaveBeenCalled() + } +) + +it('an unknown handle does not fence another worker or access the database', async () => { + const worker = await h.startSettledWorker() + h.runtime.registerPreAllocatedHandleForPty('pty-worker', 'term_worker') + h.runtime.registerPty('pty-worker', 'repo::worktree', undefined, { + tabId: 'tab_worker', + leafId: 'bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' + }) + vi.mocked(h.runtime.getTerminalPaneKey).mockRestore() + const db = vi.spyOn(h.runtime, 'getOrchestrationDb') + await expect( + h.call('orchestration.workerTerminalUserInput', { terminal: 'term_missing' }) + ).resolves.toEqual({ changed: 0 }) + expect(db).not.toHaveBeenCalled() + expect(h.db.getWorkerTerminalResourceByOwner(worker.dispatchId)?.ownership_state).toBe('owned') + await expect( + h.call('orchestration.workerRelease', { dispatch: worker.dispatchId }) + ).resolves.toMatchObject({ state: 'released' }) +}) + +it.each(['unary', 'stream'])('mobile %s bytes do no orchestration database work', async (lane) => { + const worker = await h.startSettledWorker() + const runtime = h.runtime + runtime.registerPreAllocatedHandleForPty('pty-worker', 'term_worker') + runtime.registerPty('pty-worker', 'repo::worktree', undefined, { + tabId: 'tab_worker', + leafId: 'bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb', + incarnationId: 'runtime_test:term_worker:1' + }) + const write = vi.fn(() => true) + runtime.setPtyController({ write, kill: () => true, getForegroundProcess: async () => null }) + const commit = vi.fn(async () => {}) + vi.spyOn(runtime, 'beginMobileInputFloor').mockReturnValue({ commit, rollback: vi.fn() }) + const dbAccess = vi.spyOn(runtime, 'getOrchestrationDb') + const takeover = vi.spyOn(h.db, 'markWorkerTerminalUserOwned') + const prepare = vi.spyOn(h.db.db, 'prepare') + const exec = vi.spyOn(h.db.db, 'exec') + const params = { + terminal: 'term_worker', + text: 'x', + client: { id: 'phone', type: 'mobile' as const } + } + if (lane === 'stream') { + await expect(sendTerminalStreamInput(runtime, { ...params, isMobile: true })).resolves.toBe( + 'delivered' + ) + } else { + const method = TERMINAL_SEND_METHODS.find( + (m): m is RpcMethod => m.name === 'terminal.send' && !isStreamingMethod(m) + )! + await expect( + method.handler(method.params!.parse(params) as never, { runtime } as never) + ).resolves.toMatchObject({ send: { accepted: true } }) + } + expect(write).toHaveBeenCalledWith('pty-worker', 'x') + expect(commit).toHaveBeenCalledTimes(1) + expect(dbAccess).not.toHaveBeenCalled() + expect(takeover).not.toHaveBeenCalled() + expect(prepare).not.toHaveBeenCalled() + expect(exec).not.toHaveBeenCalled() + expect(h.db.getWorkerTerminalResourceByOwner(worker.dispatchId)?.ownership_state).toBe('owned') +}) + +it('the report is reachable from a mobile-scoped device token', async () => { + // Why: mobile tokens are gated by an allowlist before dispatch. The phone reporter swallows a + // refusal, so a missing entry silently reverts every phone to the unfenced behaviour. + const { MOBILE_RPC_METHOD_ALLOWLIST } = + await import('../../../../runtime-rpc/runtime-rpc-mobile-method-allowlist') + expect(MOBILE_RPC_METHOD_ALLOWLIST.has('orchestration.workerTerminalUserInput')).toBe(true) +}) + +// Round-1 regression (#19337 review): a phone key landing inside the worker's boot wait used to +// find no `owned` row, report `changed: 0`, and still arm the client's 30 s gate — so the real +// takeover was suppressed and `worker-release` closed the pane. #19608 writes custody at terminal +// creation, so the boot-wait key itself takes the pane. +it('a phone report during the boot wait takes the pane and fences the later release', async () => { + const gate = h.deferred() + vi.spyOn(h.runtime, 'waitForTerminal').mockReturnValue(gate.promise as never) + const task = h.db.createTask({ spec: 'mid-boot phone takeover', runId: h.activeRunId }) + const start = h.call('orchestration.workerStart', { + task: task.id, + from: 'term_coord', + agent: 'codex' + }) + await vi.waitFor(() => expect(h.runtime.waitForTerminal).toHaveBeenCalled()) + const dispatchId = ( + h.db.db.prepare("SELECT dispatch_id FROM worker_dispatches WHERE state = 'starting'").get() as { + dispatch_id: string + } + ).dispatch_id + + h.runtime.registerPreAllocatedHandleForPty('pty-worker', 'term_worker') + h.runtime.registerPty('pty-worker', 'repo::worktree', undefined, { + tabId: 'tab_worker', + leafId: 'bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' + }) + vi.mocked(h.runtime.getTerminalPaneKey).mockRestore() + await expect( + h.call('orchestration.workerTerminalUserInput', { terminal: 'term_worker' }) + ).resolves.toEqual({ changed: 1 }) + + gate.resolve({ + handle: 'term_worker', + condition: 'tui-idle', + satisfied: true, + status: 'running', + exitCode: null + }) + await expect(start).resolves.toMatchObject({ state: 'ready' }) + expect(h.db.getWorkerTerminalResourceByOwner(dispatchId)?.ownership_state).toBe('user_owned') + + h.settle(task.id, dispatchId, 'succeeded') + await expect( + h.call('orchestration.workerRelease', { dispatch: dispatchId }) + ).resolves.toMatchObject({ state: 'retained', reason: 'user_takeover', processAction: 'none' }) + expect(h.runtime.closeTerminal).not.toHaveBeenCalled() +}) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts index 5207b83c8de..e6466ed48c4 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts @@ -322,18 +322,36 @@ describe('orchestration worker release', () => { expect(h.db.getWorkerTerminalResourceByOwner(dispatchId)?.ownership_state).toBe('user_owned') }) - it('retains when the exact process identity changed instead of closing', async () => { + it('keeps a resumed settled worker retained in worker-list without re-dispatch or release', async () => { h.setup() - const { dispatchId } = await h.startSettledWorker() + const { dispatchId, taskId } = await h.startSettledWorker() + const dispatch = h.db.getDispatchContextById(dispatchId) + const task = h.db.getTask(taskId) + vi.mocked(h.runtime.createTerminal).mockClear() + vi.mocked(h.runtime.sendTerminalAgentPrompt).mockClear() vi.mocked(h.runtime.getTerminalProcessIncarnation).mockImplementation((handle) => handle === 'term_worker' ? 'runtime_test:term_worker:2' : null ) - const receipt = (await h.call('orchestration.workerRelease', { dispatch: dispatchId })) as { - state: string - reason?: string + + await expect( + h.call('orchestration.workerRelease', { dispatch: dispatchId }) + ).resolves.toMatchObject({ + state: 'retained', + reason: 'identity_unproven', + processAction: 'none' + }) + const listed = (await h.call('orchestration.workerList', { run: h.activeRunId })) as { + workers: { dispatchId: string; terminalState: string; workerState: string }[] } - expect(receipt).toMatchObject({ state: 'retained', reason: 'identity_unproven' }) + expect(listed.workers).toEqual([ + expect.objectContaining({ dispatchId, terminalState: 'retained', workerState: 'succeeded' }) + ]) + expect(h.db.getTask(taskId)).toEqual(task) + expect(h.db.getDispatchContextById(dispatchId)).toEqual(dispatch) + expect(h.db.getWorkerTerminalResourceByOwner(dispatchId)?.release_state).toBe('retained') expect(h.runtime.closeTerminal).not.toHaveBeenCalled() + expect(h.runtime.createTerminal).not.toHaveBeenCalled() + expect(h.runtime.sendTerminalAgentPrompt).not.toHaveBeenCalled() }) it('retains when the terminal host scope changed instead of closing', async () => { diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts index 2a24a3efd0e..e121f1b3f25 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts @@ -10,7 +10,6 @@ import { type WorkerReleaseReceipt } from './worker-release-completion' import { WorkerDispatchParams, WorkerRetainParams } from './worker-release-schemas' -import { sweepSettledWorkerResumeFences } from '../../settled-worker-resume-fence-sweep' export const ORCHESTRATION_WORKER_RELEASE_METHODS: RpcMethod[] = [ defineMethod({ @@ -137,22 +136,27 @@ export const ORCHESTRATION_WORKER_RELEASE_METHODS: RpcMethod[] = [ // identity credential that never leaves main, so the caller names the session and the owning // runtime resolves it — a renderer echoing the pane key back would make it learnable. params: z - .object({ paneKey: z.string().min(1).optional(), sessionId: z.string().min(1).optional() }) - .refine((value) => Boolean(value.paneKey ?? value.sessionId), 'Missing paneKey or sessionId'), + .object({ + paneKey: z.string().min(1).optional(), + sessionId: z.string().min(1).optional(), + terminal: z.string().min(1).optional() + }) + .refine( + (value) => Boolean(value.paneKey ?? value.sessionId ?? value.terminal), + 'Missing paneKey, sessionId or terminal' + ), // Real user keystrokes durably relinquish orchestration ownership on the owning runtime, so // restarts, SSH drops, remote viewing, and renderer remounts cannot erase the takeover. handler: (params, { runtime }) => { // A structured worker reports by session id; it has no pane of its own to name. const paneKey = - params.paneKey ?? runtime.getStructuredWorkerPaneKeyForSession(params.sessionId!) + params.paneKey ?? + (params.sessionId + ? runtime.getStructuredWorkerPaneKeyForSession(params.sessionId) + : runtime.getTerminalPaneKey(params.terminal!)) const changed = paneKey ? runtime.getOrchestrationDb().markWorkerTerminalUserOwned(paneKey) : 0 - if (changed > 0) { - // Only a real takeover retires the resource; ordinary panes report here too and must not - // pay for a plan read on every keystroke window. - sweepSettledWorkerResumeFences(runtime) - } return { changed } } }) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-setup-gate.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-setup-gate.ts index 98407cdb1b4..52a70433867 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-setup-gate.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-setup-gate.ts @@ -6,6 +6,8 @@ import { } from './worker-topology' function residualWorkerEffects(effects: WorkerEffect[]): WorkerEffect[] { + // 'reused_agent_terminal' is the retired verb agent-first creation used for its own agent + // terminal; rows persisted before the rename still carry it. return effects.filter( (effect) => effect.action?.startsWith('created') || effect.action === 'reused_agent_terminal' ) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-agent-placement.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-agent-placement.ts new file mode 100644 index 00000000000..8d87ea894c8 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-agent-placement.ts @@ -0,0 +1,195 @@ +/** + * Where a worker's agent comes from: a worktree this start creates, a structured session, a new + * terminal in an existing worktree, or the terminal the caller passed. + * + * A structured worker never takes the agent-first worktree path. `createWorkerWorktree` used to be + * the only way a new worktree was made, and it creates one WITH its startup agent terminal, which + * left the structured branch below it unreachable for every `--worktree new-child` dispatch. Here + * the worktree is created without a startup agent and the structured session is created for it + * afterwards — the same order the renderer's own structured worktree create uses. + * + * That reorder is also why the host verdict lands here: `agentSession.createSupport` can only be + * asked about a workspace that exists, so for a created worktree it cannot run before the start. + * A refusal becomes a terminal agent in the worktree that was just created, never a failed start. + */ + +import type { AgentLaunchPreferences } from '../../../../../../shared/agent-session-host-authority' +import type { TuiAgent } from '../../../../../../shared/tui-agent' +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import type { OrchestrationDb } from '../../../../orchestration/db' +import { + resolveWorkerStartModeOnHost, + type WorkerStartModeReceipt +} from '../../orchestration-worker-start-mode' +import type { WorkerStartInput } from './worker-start-schema' +import { + createExistingWorktreeWorkerTerminal, + createStructuredWorkerSessionForWorktree, + type WorkerEffect, + type WorkerSetupReceipt +} from './worker-topology' +import { createWorkerWorktree } from './worker-worktree-creation' + +/** Only what the placement itself reads. The runtime's own worktree accessors are untyped, so + * naming the two fields keeps `any` out of this module's unions. */ +type PlacedWorktree = { id: string; repoId: string } +type WorkerStructuredSession = Awaited> + +export type WorkerAgentPlacement = { + /** The mode that actually ran; a created worktree can settle it later than the caller could. */ + mode: WorkerStartModeReceipt + worktree: PlacedWorktree + terminalHandle: string + structuredSession: WorkerStructuredSession | null + setupReceipt: WorkerSetupReceipt + warning?: string +} + +type WorkerAgentPlacementArgs = { + runtime: OrcaRuntimeService + db: OrchestrationDb + dispatchId: string + taskId: string + params: WorkerStartInput + requestedWorktree: string + /** The coordinator's worktree, present only when this start creates one. */ + creationWorktree: PlacedWorktree | undefined + /** The already-resolved placement, present only when this start does not create one. */ + resolvedWorktree: PlacedWorktree | undefined + mode: WorkerStartModeReceipt + agent: TuiAgent | undefined + launchPreferences: AgentLaunchPreferences | undefined + effects: WorkerEffect[] + /** Attributes a throw to the step that was running, the way the caller's own stages do. */ + onStage: (stage: string) => void +} + +/** The setup receipt for a placement that creates no worktree, and the one a start reports if it + * fails before a placement exists. */ +export const EXISTING_WORKTREE_SETUP: WorkerSetupReceipt = { + requested: 'not_applicable', + effective: 'not_applicable', + source: 'existing_worktree', + hookFound: false, + startupPolicy: 'start-immediately', + state: 'not_applicable' +} + +export async function placeWorkerAgent( + args: WorkerAgentPlacementArgs +): Promise { + if (args.creationWorktree) { + return placeInCreatedWorktree(args, args.creationWorktree) + } + const worktree = requireWorktree(args.resolvedWorktree) + if (args.params.terminal) { + args.effects.push({ + kind: 'terminal', + role: 'agent', + action: 'reused', + id: args.params.terminal + }) + return { + mode: args.mode, + worktree, + terminalHandle: args.params.terminal, + structuredSession: null, + setupReceipt: EXISTING_WORKTREE_SETUP + } + } + return { + mode: args.mode, + worktree, + ...(await createWorkerAgentSurface(args, worktree.id, args.mode)), + setupReceipt: EXISTING_WORKTREE_SETUP + } +} + +async function placeInCreatedWorktree( + args: WorkerAgentPlacementArgs, + coordinatorWorktree: PlacedWorktree +): Promise { + args.onStage('worktree_create') + const created = await createWorkerWorktree({ + runtime: args.runtime, + db: args.db, + dispatchId: args.dispatchId, + requestedWorktree: args.requestedWorktree, + coordinatorWorktree, + params: args.params, + agent: args.agent as TuiAgent, + withAgentTerminal: args.mode.mode !== 'structured', + ...(args.launchPreferences ? { launchPreferences: args.launchPreferences } : {}), + effects: args.effects + }) + const worktree = requireWorktree(created.worktree) + if (args.mode.mode !== 'structured') { + return { + mode: args.mode, + worktree, + terminalHandle: requireTerminal(created.terminalHandle), + structuredSession: null, + setupReceipt: created.setupReceipt + } + } + args.onStage('terminal_create') + const mode = await resolveWorkerStartModeOnHost(args.runtime, args.mode, worktree.id, args.agent) + return { + mode, + worktree, + ...(await createWorkerAgentSurface(args, worktree.id, mode)), + setupReceipt: created.setupReceipt + } +} + +/** The agent surface for a worktree that exists; the settled mode picks which one. */ +async function createWorkerAgentSurface( + args: WorkerAgentPlacementArgs, + worktreeId: string, + mode: WorkerStartModeReceipt +): Promise> { + args.db.recordWorkerStage({ + dispatchId: args.dispatchId, + stage: 'terminal_creating', + worktreeId, + effects: args.effects + }) + if (mode.mode === 'structured') { + const structuredSession = await createStructuredWorkerSessionForWorktree({ + runtime: args.runtime, + worktreeId, + agent: args.agent as TuiAgent, + dispatchId: args.dispatchId, + ...(args.launchPreferences ? { launchPreferences: args.launchPreferences } : {}), + effects: args.effects + }) + return { terminalHandle: structuredSession.identity.handle, structuredSession } + } + const terminal = await createExistingWorktreeWorkerTerminal({ + runtime: args.runtime, + worktreeId, + agent: args.agent as TuiAgent, + ...(args.launchPreferences ? { launchPreferences: args.launchPreferences } : {}), + taskId: args.taskId, + effects: args.effects + }) + return { + terminalHandle: terminal.handle, + structuredSession: null, + ...(terminal.warning ? { warning: terminal.warning } : {}) + } +} + +function requireWorktree(worktree: PlacedWorktree | undefined): PlacedWorktree { + if (!worktree) { + throw new Error('Worker topology did not resolve a worktree.') + } + return worktree +} + +function requireTerminal(terminalHandle: string | undefined): string { + if (!terminalHandle) { + throw new Error('Worker topology did not resolve an agent terminal.') + } + return terminalHandle +} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts index dc645c97c88..593d097580f 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts @@ -41,6 +41,8 @@ const openDatabases: OrchestrationDb[] = [] const temporaryRoots: string[] = [] type PromptContractHarness = { + runtime: Awaited>['runtime'] + handle: string db: OrchestrationDb dbPath: string dispatcher: RpcDispatcher @@ -131,6 +133,8 @@ async function createPromptContractHarness( vi.spyOn(runtime, 'getTerminalOrchestrationCliCommand').mockReturnValue('orca') return { + runtime, + handle, db, dbPath, dispatcher: new RpcDispatcher({ runtime, methods: ORCHESTRATION_METHODS }), @@ -227,22 +231,100 @@ describe('orchestration worker-start prompt contract', () => { }) }) - it('keeps a swallowed Enter queued without revoking the worker or retrying input', async () => { + it.each([ + ['succeeded', false], + ['succeeded', true], + ['failed', false], + ['failed', true] + ] as const)('preserves an early %s report with turn evidence=%s', async (outcome, observed) => { + vi.useFakeTimers() + const harness = await createPromptContractHarness('swallowed') + vi.spyOn(harness.runtime, 'observeTerminalAgentPrompt').mockImplementation( + async (_handle, prompt) => { + const dispatch = harness.db.findActiveDispatchForAssignee(harness.handle) + expect(dispatch).toBeDefined() + expect( + harness.db.settleWorkerReport({ + taskId: harness.taskId, + dispatchId: dispatch!.id, + outcome, + result: 'Finished before the hook arrived' + }) + ).toMatchObject({ action: 'settled', outcome }) + return observed ? { ...prompt, stages: ['input_accepted', 'turn_started'] } : prompt + } + ) + const pending = harness.dispatcher.dispatch(harness.request) + await vi.runAllTimersAsync() + expect(await pending).toMatchObject({ + ok: true, + result: { state: 'ready', stage: 'settled', workerOutcome: outcome } + }) + expect(harness.db.getTask(harness.taskId)?.status).toBe( + outcome === 'succeeded' ? 'completed' : 'failed' + ) + }) + + it('retains accepted authority when the observation binding becomes stale', async () => { + vi.useFakeTimers() + const harness = await createPromptContractHarness('swallowed') + vi.spyOn(harness.runtime, 'observeTerminalAgentPrompt').mockRejectedValue( + new Error('terminal_handle_stale') + ) + const pending = harness.dispatcher.dispatch(harness.request) + await vi.runAllTimersAsync() + expect(await pending).toMatchObject({ ok: true, result: { state: 'outcome_unknown' } }) + expect(harness.db.findActiveDispatchForAssignee(harness.handle)).toMatchObject({ + status: 'pending', + capability_hash: expect.any(String), + capability_revoked_at: null + }) + expect(harness.submittedTurns()).toBe(1) + expect(vi.getTimerCount()).toBe(0) + }) + + it('keeps a worker question answerable after turn observation times out', async () => { + vi.useFakeTimers() + const harness = await createPromptContractHarness('swallowed') + let questionId = '' + vi.spyOn(harness.runtime, 'observeTerminalAgentPrompt').mockImplementation( + async (_handle, prompt) => { + const dispatch = harness.db.findActiveDispatchForAssignee(harness.handle)! + questionId = harness.db.createQuestion({ + runId: dispatch.run_id!, + dispatchId: dispatch.id, + askerHandle: harness.handle, + question: 'Which target should I use?' + }).question.message_id + return prompt + } + ) + const pending = harness.dispatcher.dispatch(harness.request) + await vi.runAllTimersAsync() + expect(await pending).toMatchObject({ ok: true, result: { state: 'outcome_unknown' } }) + expect(harness.db.getQuestion(questionId)?.status).toBe('pending') + }) + + it('reports a swallowed Enter as start_unknown while keeping the worker and its capability', async () => { vi.useFakeTimers() const harness = await createPromptContractHarness('swallowed') const pending = harness.dispatcher.dispatch(harness.request) await vi.runAllTimersAsync() const response = await pending + // Codex supports turn-start observation and no turn started, so ready would be a lie: the + // paste can sit unsent in the composer while the receipt looks like a healthy dispatch. expect(response).toMatchObject({ ok: true, result: { - state: 'ready', - stage: 'input_accepted', + state: 'outcome_unknown', + stage: 'turn_start_unobserved', + turnStart: 'unobserved', prompt: { requestId: harness.requestId, stages: ['input_accepted'] }, + nextCommands: expect.arrayContaining([expect.stringContaining('worker-show')]), mutation: { requestId: harness.requestId, replayed: false } } }) @@ -251,29 +333,42 @@ describe('orchestration worker-start prompt contract', () => { } const dispatchId = (response.result as { dispatchId: string }).dispatchId await vi.advanceTimersByTimeAsync(20_000) + // Unverifiable is not failure: exactly one submit, no blind retry, nothing torn down. expect(harness.submittedTurns()).toBe(1) expect(harness.startedTurns()).toBe(0) expect(harness.prematureSubmits()).toBe(0) expect(harness.writes.filter((data) => data === '\r')).toHaveLength(1) const persisted = reopenPromptContractDb(harness) - expect(persisted.getTask(harness.taskId)?.status).toBe('dispatched') + expect(persisted.getTask(harness.taskId)?.status).toBe('blocked') expect(persisted.getDispatchContextById(dispatchId)).toMatchObject({ - status: 'dispatched', + status: 'pending', last_failure: null, + // The capability survives so a worker that recovers can still report; worker-report + // settlement reconnects a start_unknown worker through 'ready'. + capability_hash: expect.any(String), capability_revoked_at: null }) expect(persisted.getWorkerDispatch(dispatchId)).toMatchObject({ - state: 'ready', - stage: 'input_accepted', - last_error: null + state: 'start_unknown', + stage: 'turn_start_unobserved', + last_error: expect.stringContaining('turn start could not be verified') }) + const persistedEffects = JSON.parse( + persisted.getWorkerDispatch(dispatchId)?.effects ?? '[]' + ) as { kind?: string; state?: string }[] + expect(persistedEffects).toEqual( + expect.arrayContaining([ + expect.objectContaining({ kind: 'dispatch_input', state: 'accepted' }), + expect.objectContaining({ kind: 'dispatch_input', state: 'turn_unobserved' }) + ]) + ) const callerFingerprint = persisted.getOrCreateLocalMutationCallerFingerprint() const receipt = persisted.getMutationReceipt(callerFingerprint, harness.requestId) expect(receipt).toMatchObject({ state: 'completed' }) expect(JSON.parse(receipt?.receipt ?? 'null')).toMatchObject({ dispatchId, - state: 'ready', - stage: 'input_accepted', + state: 'outcome_unknown', + stage: 'turn_start_unobserved', prompt: { requestId: harness.requestId, stages: ['input_accepted'] diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-readiness-settlement.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-readiness-settlement.ts new file mode 100644 index 00000000000..a3c57a357d2 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-readiness-settlement.ts @@ -0,0 +1,156 @@ +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import type { OrchestrationDb } from '../../../../orchestration/db' +import type { RunRow, TaskRow } from '../../../../orchestration/types' +import type { WorkerStartModeReceipt } from '../../orchestration-worker-start-mode' +import { deliverWorkerDispatchPreamble } from './deliver-worker-dispatch-preamble' +import type { OrchestrationWorkerLaunchReceipt } from './worker-launch-preferences' +import { + describeUnobservedWorkerTurnStart, + observeWorkerTurnStart, + type WorkerTurnStartObservation +} from './worker-start-turn-observation' +import { + monitorWorkerSetup, + type createStructuredWorkerSessionForWorktree, + type WorkerEffect, + type WorkerSetupReceipt +} from './worker-topology' + +/** + * Delivers the dispatch preamble and settles the worker's start state on the strongest + * evidence available: `ready` only with a positive turn-start (or a provider that cannot + * prove one), `start_unknown` when observation is supported and nothing started. + */ +export async function deliverAndSettleWorkerStartReadiness(args: { + runtime: OrcaRuntimeService + db: OrchestrationDb + run: RunRow + task: TaskRow + dispatchId: string + dispatchDepth: number + structuredSession: Awaited> | null + terminalHandle: string + coordinatorHandle: string + dispatchCapability: string + devMode: boolean | undefined + requestId: string + agent: string | null + setupReceipt: WorkerSetupReceipt + launchReceipt: OrchestrationWorkerLaunchReceipt + mode: WorkerStartModeReceipt + timeoutMs: number + effects: WorkerEffect[] + terminalRevealWarning: string | undefined + /** Keeps the caller's failure receipt naming the stage that actually failed. */ + onStage: (stage: 'dispatch_input' | 'turn_observation') => void +}): Promise { + const { runtime, db, run, task, structuredSession, terminalHandle, effects } = args + + args.onStage('dispatch_input') + const promptDelivery = await deliverWorkerDispatchPreamble({ + runtime, + structuredSession, + terminalHandle, + dispatchId: args.dispatchId, + dispatchDepth: args.dispatchDepth, + taskId: task.id, + taskSpec: task.spec, + coordinatorHandle: args.coordinatorHandle, + dispatchCapability: args.dispatchCapability, + devMode: args.devMode, + requestId: args.requestId + }) + effects.push({ + kind: 'dispatch_input', + role: 'agent', + id: terminalHandle, + state: 'accepted' + }) + + args.onStage('turn_observation') + // The write above was accepted without waiting on provider hooks; now demand the positive + // evidence the receipt claims is observable. A worker whose turn never starts must not be + // reported ready — a wedged agent and a working one looked identical before this gate. + // A structured preamble send is acknowledged by the provider or throws, so it is already + // positive evidence. + const turnStart: WorkerTurnStartObservation = structuredSession + ? { verdict: 'observed' } + : await observeWorkerTurnStart({ runtime, terminalHandle, prompt: promptDelivery }) + const deliveredPrompt = turnStart.prompt ?? promptDelivery + monitorWorkerSetup({ + runtime, + db, + runId: run.id, + dispatchId: args.dispatchId, + setupReceipt: args.setupReceipt, + effects + }) + // A worker report can settle the dispatch while turn observation is outstanding. + const currentWorker = db.getWorkerDispatch(args.dispatchId) + const alreadySettled = currentWorker && currentWorker.state !== 'starting' + if (turnStart.verdict === 'unobserved' && !alreadySettled) { + // Honest `unverifiable`: keep the dispatch capability and the terminal — the worker may + // still recover and report (worker-report settlement reconnects a start_unknown worker) — + // but never claim ready for a turn nobody observed. + effects.push({ + kind: 'dispatch_input', + role: 'agent', + id: terminalHandle, + state: 'turn_unobserved' + }) + const reason = describeUnobservedWorkerTurnStart(args.agent) + const worker = db.markWorkerStartUnknown( + args.dispatchId, + 'turn_start_unobserved', + reason, + effects + ) + return { + runId: run.id, + taskId: task.id, + dispatchId: args.dispatchId, + state: 'outcome_unknown', + stage: worker.stage, + turnStart: turnStart.verdict, + lastError: reason, + setup: args.setupReceipt, + launch: args.launchReceipt, + mode: args.mode, + timeoutMs: args.timeoutMs, + effects, + ...(deliveredPrompt ? { prompt: deliveredPrompt } : {}), + residualResources: JSON.parse(worker.residual_resources) as unknown[], + nextCommands: [ + `orca orchestration worker-show --dispatch ${args.dispatchId} --json`, + `orca terminal read --terminal ${terminalHandle} --screen`, + `orca orchestration worker-abandon --dispatch ${args.dispatchId} --json` + ], + ...(args.terminalRevealWarning ? { warning: args.terminalRevealWarning } : {}) + } + } + const worker = alreadySettled + ? currentWorker + : db.markWorkerDispatchReady(args.dispatchId, effects) + // A completed task proves start succeeded; older callers use only 'ready' as start success. + const reportedOutcome = + worker.stage === 'settled' && (worker.state === 'succeeded' || worker.state === 'failed') + ? worker.state + : undefined + return { + runId: run.id, + taskId: task.id, + dispatchId: args.dispatchId, + state: reportedOutcome ? 'ready' : worker.state, + stage: worker.stage, + ...(reportedOutcome ? { workerOutcome: reportedOutcome } : {}), + turnStart: turnStart.verdict, + setup: args.setupReceipt, + launch: args.launchReceipt, + mode: args.mode, + timeoutMs: args.timeoutMs, + effects, + ...(deliveredPrompt ? { prompt: deliveredPrompt } : {}), + residualResources: [], + ...(args.terminalRevealWarning ? { warning: args.terminalRevealWarning } : {}) + } +} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-receipt.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-receipt.ts index 9fd98dd9db3..f2dee00b44c 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-receipt.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-receipt.ts @@ -4,8 +4,8 @@ import { isUnknownWorkerStartOutcome, type WorkerSetupReceipt } from './worker-t import type { OrchestrationWorkerLaunchReceipt } from './worker-launch-preferences' import type { WorkerStartModeReceipt } from '../../orchestration-worker-start-mode' import { isAgentSessionPtyWriteRefusedError } from '../../../../../../shared/agent-session-pty-write-admission' -import type { FailedStartTerminalAdoption } from '../../../../orchestration/db/worker-terminal/failed-start-terminal-adoption' import { structuredChatPtyWriteRefusalCopy } from '../../../../../../shared/agent-session-pty-write-refusal-copy' +import { isStructuredWorkerHandle } from '../../../../structured-worker-identity' export function failWorkerStartWithReceipt(args: { db: OrchestrationDb @@ -17,8 +17,6 @@ export function failWorkerStartWithReceipt(args: { setup: WorkerSetupReceipt launch: OrchestrationWorkerLaunchReceipt mode: WorkerStartModeReceipt - /** The terminal this start created and never handed to an owner. */ - residualAgentTerminal?: FailedStartTerminalAdoption }): unknown { const agentSessionRefusal = isAgentSessionPtyWriteRefusedError(args.error) ? args.error.refusal @@ -33,14 +31,14 @@ export function failWorkerStartWithReceipt(args: { : args.db.failWorkerStart(args.dispatchId, args.failedStage, reason, { // Why (#16095): the preamble is written before submission is verified, so a stalled // verdict never means the worker lacks its task — keep the authority its report needs. - retainCapability: isAgentPromptStalledError(args.error), - ...(args.residualAgentTerminal ? { adoptResidualTerminal: args.residualAgentTerminal } : {}) + retainCapability: isAgentPromptStalledError(args.error) }) - // Only claim cleanup the ownership table actually accepted; the adoption declines a terminal - // another resource already accounts for. - const adopted = - Boolean(args.residualAgentTerminal) && - Boolean(args.db.getWorkerTerminalResourceByOwner(args.dispatchId)) + // Only name cleanup this start actually left behind: a terminal it created and still owns. A + // structured session is discarded by the teardown, a pane the user typed into is theirs, and an + // unknown outcome is not settled — none of the three has anything for `worker-release` to close. + const residual = unknown ? undefined : args.db.getWorkerTerminalResourceByOwner(args.dispatchId) + const releasable = + residual?.ownership_state === 'owned' && !isStructuredWorkerHandle(residual.terminal_handle) return { runId: args.runId, taskId: args.taskId, @@ -55,7 +53,7 @@ export function failWorkerStartWithReceipt(args: { effects: JSON.parse(worker.effects) as unknown[], residualResources: JSON.parse(worker.residual_resources) as unknown[], ...(agentSessionRefusal ? { agentSessionRefusal } : {}), - ...(adopted + ...(releasable ? { recovery: `This start created a terminal that never ran the Task. Close it with: orca orchestration worker-release --dispatch ${args.dispatchId}` } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-structured-setup-gate.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-structured-setup-gate.ts new file mode 100644 index 00000000000..d4068ac6776 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-structured-setup-gate.ts @@ -0,0 +1,67 @@ +/** + * The `wait-for-setup` gate for a structured worker on a worktree this start created. + * + * A PTY worker gets the gate for free: agent-first creation sequences the agent's startup command + * behind the setup runner, so `tui-idle` cannot arrive until setup exits, and the worker start + * reads the gate's outcome off that wait. A structured session has no startup command to sequence, + * so without this the worker would take its dispatch preamble while `install` was still running, + * and the repo's wait-for-setup policy would record no evidence at all. + * + * Bounded by the start's own timeout, and deliberately forgiving: a wait that cannot be taken — + * an in-process hook with no setup terminal, or a setup pty already gone — yields no verdict + * rather than a failure, because a worker start must not fail on missing evidence. + * + * "No verdict" is never silent, though. A wait that could not be TAKEN is an absent precondition + * and needs no receipt; a wait that was taken and then threw is a LOST observation, and that one + * is recorded as a `wait_unevaluated` effect so the start's receipt still says the gate went + * unevaluated and why. + */ + +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import type { WorkerEffect, WorkerSetupReceipt } from './worker-topology' + +export type StructuredWorkerSetupGate = { + satisfied: boolean + status: string + /** A setup gate has no agent prompt to block on; declared so the wait union stays property-typed. */ + blockedReason?: undefined +} + +export async function awaitStructuredWorkerSetupGate(args: { + runtime: Pick + setup: WorkerSetupReceipt + effects: WorkerEffect[] + timeoutMs: number +}): Promise { + if (args.setup.startupPolicy !== 'wait-for-setup' || args.setup.state !== 'running') { + return null + } + const setupTerminal = args.effects.find((effect) => effect.kind === 'setup')?.terminalId + if (!setupTerminal) { + return null + } + let timer: ReturnType | undefined + try { + return await Promise.race([ + args.runtime.waitForSetupTerminalCompletion(setupTerminal).then((completion) => ({ + satisfied: completion.exitCode === 0, + status: 'exited' + })), + new Promise((resolve) => { + timer = setTimeout(() => resolve({ satisfied: false, status: 'timeout' }), args.timeoutMs) + }) + ]) + } catch (error) { + // A wait that was TAKEN and then threw is not the same as one that could not be taken. Both + // yield no verdict — a start must not fail on missing evidence — but only this one is a lost + // observation, so it is recorded rather than silently flattened into "not applicable". + args.effects.push({ + kind: 'setup', + action: 'wait_unevaluated', + state: error instanceof Error ? error.message : String(error) + }) + return null + } finally { + clearTimeout(timer) + } +} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-turn-observation.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-turn-observation.test.ts new file mode 100644 index 00000000000..f1992eb9b9c --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-turn-observation.test.ts @@ -0,0 +1,120 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RuntimeTerminalPromptDelivery } from '../../../../../../shared/runtime-terminal-contracts' +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import { observeWorkerTurnStart } from './worker-start-turn-observation' + +function delivery( + overrides: Partial = {} +): RuntimeTerminalPromptDelivery { + return { + requestId: 'req-1', + stages: ['input_accepted'], + provider: 'codex', + observation: 'supported', + processIncarnation: 'inc-1', + generation: 1, + baselineWorkingSequence: 0, + ...overrides + } +} + +function runtimeObserving(result: RuntimeTerminalPromptDelivery): { + runtime: OrcaRuntimeService + observe: ReturnType +} { + const observe = vi.fn().mockResolvedValue(result) + return { + runtime: { observeTerminalAgentPrompt: observe } as unknown as OrcaRuntimeService, + observe + } +} + +describe('observeWorkerTurnStart', () => { + it('treats a missing receipt as unsupported observation, never as failure', async () => { + const { runtime, observe } = runtimeObserving(delivery()) + await expect( + observeWorkerTurnStart({ runtime, terminalHandle: 'term_w', prompt: undefined }) + ).resolves.toEqual({ verdict: 'unsupported' }) + expect(observe).not.toHaveBeenCalled() + }) + + it('accepts a first-stage turn_started without a second observation pass', async () => { + const prompt = delivery({ stages: ['input_accepted', 'turn_started'] }) + const { runtime, observe } = runtimeObserving(prompt) + await expect( + observeWorkerTurnStart({ runtime, terminalHandle: 'term_w', prompt }) + ).resolves.toEqual({ verdict: 'observed', prompt }) + expect(observe).not.toHaveBeenCalled() + }) + + it('reports observed when the second-stage observer sees the turn start', async () => { + const observed = delivery({ stages: ['input_accepted', 'turn_started'] }) + const { runtime, observe } = runtimeObserving(observed) + await expect( + observeWorkerTurnStart({ + runtime, + terminalHandle: 'term_w', + prompt: delivery(), + timeoutMs: 5 + }) + ).resolves.toEqual({ verdict: 'observed', prompt: observed }) + expect(observe).toHaveBeenCalledWith('term_w', delivery(), 5) + }) + + it('reports unobserved — not dead — when a supported observation stalls', async () => { + const { runtime } = runtimeObserving(delivery()) + await expect( + observeWorkerTurnStart({ + runtime, + terminalHandle: 'term_w', + prompt: delivery(), + timeoutMs: 5 + }) + ).resolves.toMatchObject({ verdict: 'unobserved' }) + }) + + it('reports permission as positive liveness', async () => { + const observed = delivery({ observation: 'permission' }) + const { runtime } = runtimeObserving(observed) + await expect( + observeWorkerTurnStart({ + runtime, + terminalHandle: 'term_w', + prompt: delivery(), + timeoutMs: 5 + }) + ).resolves.toEqual({ verdict: 'permission', prompt: observed }) + }) + + it('treats a replaced incarnation as unobserved rather than unsupported', async () => { + const observed = delivery({ observation: 'incarnation_replaced' }) + const { runtime } = runtimeObserving(observed) + await expect( + observeWorkerTurnStart({ + runtime, + terminalHandle: 'term_w', + prompt: delivery(), + timeoutMs: 5 + }) + ).resolves.toEqual({ verdict: 'unobserved', prompt: observed }) + }) + + it('leaves an unsupported provider on the accepted receipt', async () => { + const prompt = delivery({ provider: 'unsupported', observation: 'unsupported' }) + const { runtime, observe } = runtimeObserving(prompt) + await expect( + observeWorkerTurnStart({ runtime, terminalHandle: 'term_w', prompt }) + ).resolves.toEqual({ verdict: 'unsupported', prompt }) + expect(observe).not.toHaveBeenCalled() + }) + + it('preserves uncertainty when observation loses the terminal binding', async () => { + const prompt = delivery() + const { runtime, observe } = runtimeObserving(prompt) + observe.mockRejectedValue(new Error('terminal_handle_stale')) + await expect( + observeWorkerTurnStart({ runtime, terminalHandle: 'term_w', prompt }) + ).resolves.toEqual({ verdict: 'unobserved', prompt }) + expect(observe).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-turn-observation.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-turn-observation.ts new file mode 100644 index 00000000000..d1c9e59adfa --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-turn-observation.ts @@ -0,0 +1,86 @@ +import { AGENT_PROMPT_EFFECT_TIMEOUT_MS } from '../../../../../../shared/orchestration-timing-budgets' +import type { RuntimeTerminalPromptDelivery } from '../../../../../../shared/runtime-terminal-contracts' +import type { OrcaRuntimeService } from '../../../../orca-runtime' + +/** + * Turn-start verdict for a dispatched worker prompt, in the execution-boundary vocabulary: + * + * - 'observed': the provider proved a turn started for this request. Positive liveness. + * - 'permission': the agent rendered an approval prompt after the write. Positive liveness, + * but the turn is blocked on a human. + * - 'unsupported': this provider exposes no turn-start signal; the accepted write is the + * strongest receipt that can exist. Never treated as failure. + * - 'unobserved': observation IS supported and no turn started within the window. This is + * `unverifiable`, never evidence of death — the bytes were written, but the agent may be + * wedged at startup or holding the spec unsent in its composer. + */ +export type WorkerTurnStartVerdict = 'observed' | 'permission' | 'unsupported' | 'unobserved' + +export type WorkerTurnStartObservation = { + verdict: WorkerTurnStartVerdict + prompt?: RuntimeTerminalPromptDelivery +} + +function classifyPromptDelivery(prompt: RuntimeTerminalPromptDelivery): WorkerTurnStartVerdict { + if (prompt.stages.includes('turn_started')) { + return 'observed' + } + if (prompt.observation === 'permission') { + return 'permission' + } + if (prompt.observation === 'supported') { + return 'unobserved' + } + // 'unsupported' (and an old host's missing observation) leaves acceptance as the best receipt. + return 'unsupported' +} + +/** + * Second-stage turn-start observation for a worker prompt that was accepted without waiting on + * provider hooks. Reuses the same observer that terminal.send receipts replay through, so the + * evidence rules (lifecycle edge or hook turn-start, never output bytes) stay in one place. + * + * The observation window is `AGENT_PROMPT_EFFECT_TIMEOUT_MS`, which worker-start's client RPC + * grace already budgets for (see orchestration-worker-start-prompt-budget.ts). + */ +export async function observeWorkerTurnStart(args: { + runtime: OrcaRuntimeService + terminalHandle: string + prompt: RuntimeTerminalPromptDelivery | undefined + timeoutMs?: number +}): Promise { + if (!args.prompt) { + return { verdict: 'unsupported' } + } + const verdict = classifyPromptDelivery(args.prompt) + if (verdict !== 'unobserved') { + return { verdict, prompt: args.prompt } + } + let observed: RuntimeTerminalPromptDelivery + try { + observed = await args.runtime.observeTerminalAgentPrompt( + args.terminalHandle, + args.prompt, + args.timeoutMs ?? AGENT_PROMPT_EFFECT_TIMEOUT_MS + ) + } catch { + // Observation failure cannot revoke authority for input that was already accepted. + return { verdict: 'unobserved', prompt: args.prompt } + } + if (observed.observation === 'incarnation_replaced') { + // The PTY under this handle changed mid-observation; the accepted write is unproven. + return { verdict: 'unobserved', prompt: observed } + } + return { verdict: classifyPromptDelivery(observed), prompt: observed } +} + +export function describeUnobservedWorkerTurnStart(agent: string | null): string { + const name = agent ?? 'the agent' + return ( + `Dispatch input was written and submitted, but ${name}'s turn start could not be verified ` + + `during observation (up to ${Math.round(AGENT_PROMPT_EFFECT_TIMEOUT_MS / 1000)}s). This is unverifiable, not proof the ` + + 'worker is dead: the agent may still be starting, may be wedged (for example waiting on ' + + 'network), or may be holding the task unsent in its composer. If the worker recovers and ' + + 'reports, this Dispatch settles normally.' + ) +} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts index 605d8c52d4a..643323cf62e 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts @@ -245,8 +245,9 @@ export const ORCHESTRATION_WORKER_STOP_METHODS: RpcMethod[] = [ const activeStopByRuntime = new WeakMap>>() -/** Two callers stopping one Dispatch: the second reached `beginWorkerStop` after the first moved - * the row to `stopping` and got `dispatch_inactive` instead of the first caller's receipt. */ +/** Two callers stopping one Dispatch: coalesced so only one of them closes the terminal. Both are + * in this runtime and so carry one epoch, which `beginWorkerStop` refuses a second time anyway; + * the epoch it does accept belongs to a row a dead runtime stranded, and no caller here holds one. */ function dedupeWorkerStop( runtime: OrcaRuntimeService, dispatchId: string, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-terminal-custody-at-creation.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-terminal-custody-at-creation.test.ts new file mode 100644 index 00000000000..201201e5877 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-terminal-custody-at-creation.test.ts @@ -0,0 +1,216 @@ +/** + * Custody for an agent terminal this start created is written when the terminal is created, not + * after the agent boot wait. + * + * A worker pane is visible on desktop and phone the moment it exists. While the row was written + * only after `tui-idle` (up to 60 s later), a keystroke into the booting pane found no `owned` row, + * `markWorkerTerminalUserOwned` returned 0, and the takeover was lost — so a later `worker-release` + * closed the pane the user had claimed. + */ + +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrchestrationDb } from '../../../../orchestration/db' +import { createOrchestrationWorkerReleaseHarness } from './worker-release.test-support' + +const READY_WAIT = { + handle: 'term_worker', + condition: 'tui-idle', + satisfied: true, + status: 'running', + exitCode: null +} + +describe('worker terminal custody is recorded at terminal creation', () => { + const h = createOrchestrationWorkerReleaseHarness() + + afterEach(() => h.cleanup()) + + /** Holds the agent boot wait open so the mid-start database state can be read. */ + function holdBootWait(): { finish: (satisfied?: boolean) => void } { + const gate = h.deferred() + vi.spyOn(h.runtime, 'waitForTerminal').mockReturnValue(gate.promise as never) + return { + finish: (satisfied = true) => + gate.resolve({ ...READY_WAIT, satisfied, status: satisfied ? 'running' : 'exited' }) + } + } + + function startingDispatchId(): string { + return ( + h.db.db + .prepare("SELECT dispatch_id FROM worker_dispatches WHERE state = 'starting'") + .get() as { dispatch_id: string } + ).dispatch_id + } + + async function startHeldAtBootWait(options: { terminal?: string } = {}): Promise<{ + dispatchId: string + taskId: string + start: Promise + finish: (satisfied?: boolean) => void + }> { + const task = h.db.createTask({ spec: 'custody at creation', runId: h.activeRunId }) + const { finish } = holdBootWait() + const start = h.call('orchestration.workerStart', { + task: task.id, + from: 'term_coord', + ...(options.terminal ? { terminal: options.terminal } : { agent: 'codex' }) + }) + await vi.waitFor(() => expect(h.runtime.waitForTerminal).toHaveBeenCalled()) + return { dispatchId: startingDispatchId(), taskId: task.id, start, finish } + } + + it('owns the created terminal before the boot wait resolves', async () => { + h.setup() + const held = await startHeldAtBootWait() + + expect(h.db.getWorkerTerminalResourceByOwner(held.dispatchId)).toMatchObject({ + ownership_state: 'owned', + release_state: 'not_requested', + terminal_handle: 'term_worker', + pane_key: h.workerPaneKey, + process_incarnation: 'runtime_test:term_worker:1', + host_scope: JSON.stringify({ kind: 'local', hostId: 'local' }) + }) + // worker-list reads the same row: a booting worker now says `active`, not `retained`. + expect(h.db.listWorkerTerminalResources({ dispatchIds: [held.dispatchId] })[0]).toMatchObject({ + agentTerminalHandle: 'term_worker', + terminalState: 'active' + }) + + held.finish() + await expect(held.start).resolves.toMatchObject({ state: 'ready' }) + }) + + it('claims nothing for an explicitly reused terminal until authority transfers it', async () => { + h.setup() + const held = await startHeldAtBootWait({ terminal: 'term_worker' }) + + expect(h.db.getWorkerTerminalResourceByOwner(held.dispatchId)).toBeUndefined() + + held.finish() + await expect(held.start).resolves.toMatchObject({ state: 'ready' }) + expect(h.db.getWorkerTerminalResourceByOwner(held.dispatchId)).toMatchObject({ + ownership_state: 'external', + retained_reason: 'external_terminal' + }) + }) + + it('lets a keystroke during the boot wait take the pane, and release then retains it', async () => { + h.setup() + const held = await startHeldAtBootWait() + + await expect( + h.call('orchestration.workerTerminalUserInput', { paneKey: h.workerPaneKey }) + ).resolves.toEqual({ changed: 1 }) + + held.finish() + await expect(held.start).resolves.toMatchObject({ state: 'ready' }) + expect(h.db.getWorkerTerminalResourceByOwner(held.dispatchId)).toMatchObject({ + ownership_state: 'user_owned', + retained_reason: 'user_takeover' + }) + + h.settle(held.taskId, held.dispatchId, 'succeeded') + await expect( + h.call('orchestration.workerRelease', { dispatch: held.dispatchId }) + ).resolves.toMatchObject({ state: 'retained', reason: 'user_takeover', processAction: 'none' }) + expect(h.runtime.closeTerminal).not.toHaveBeenCalled() + }) + + it('still refuses to release a starting worker that already owns its terminal', async () => { + h.setup() + const held = await startHeldAtBootWait() + + await expect( + h.call('orchestration.workerRelease', { dispatch: held.dispatchId }) + ).rejects.toThrow(/only a settled worker can release/) + + held.finish() + await held.start + }) + + it('leaves a start that died on the boot wait a terminal worker-release can close', async () => { + h.setup() + const held = await startHeldAtBootWait() + held.finish(false) + + await expect(held.start).resolves.toMatchObject({ + state: 'failed', + failedStage: 'agent_readiness', + recovery: expect.stringContaining('worker-release') + }) + expect(h.db.getWorkerTerminalResourceByOwner(held.dispatchId)).toMatchObject({ + ownership_state: 'owned', + terminal_handle: 'term_worker' + }) + + await expect( + h.call('orchestration.workerRelease', { dispatch: held.dispatchId }) + ).resolves.toMatchObject({ state: 'released', processAction: 'closed_agent_terminal' }) + expect(h.runtime.closeTerminal).toHaveBeenCalledWith('term_worker') + }) + + it('promises no cleanup while the start outcome is still unknown', async () => { + h.setup() + const task = h.db.createTask({ spec: 'unknown outcome', runId: h.activeRunId }) + const unknown = Object.assign(new Error('the execution host went away'), { + code: 'operation_unknown' + }) + vi.spyOn(h.runtime, 'waitForTerminal').mockRejectedValue(unknown) + + const receipt = (await h.call('orchestration.workerStart', { + task: task.id, + from: 'term_coord', + agent: 'codex' + })) as { state: string; dispatchId: string; nextCommands?: string[] } + + expect(receipt).toMatchObject({ state: 'outcome_unknown' }) + // worker-release refuses an unsettled worker, so the receipt must not name it. + expect(receipt).not.toHaveProperty('recovery') + expect(receipt.nextCommands?.join(' ')).toContain('worker-abandon') + expect(h.db.getWorkerTerminalResourceByOwner(receipt.dispatchId)).toMatchObject({ + ownership_state: 'owned' + }) + }) + + it('promises no cleanup for a reused terminal whose start died', async () => { + h.setup() + const held = await startHeldAtBootWait({ terminal: 'term_worker' }) + held.finish(false) + + const receipt = await held.start + expect(receipt).toMatchObject({ state: 'failed' }) + expect(receipt).not.toHaveProperty('recovery') + expect(h.db.getWorkerTerminalResourceByOwner(held.dispatchId)).toBeUndefined() + }) +}) + +describe('custody refuses a dispatch that stopped while its terminal was being created', () => { + let db: OrchestrationDb | undefined + + afterEach(() => db?.close()) + + it('records no owner once the dispatch is no longer starting', () => { + const d = (db = new OrchestrationDb(':memory:')) + const started = d.createStartingWorkerDispatch({ + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + taskId: d.createTask({ runId: 'run_legacy_local', spec: 'stopped mid-create' }).id, + startOptions: {} + }) + // Startup reconciliation abandons a `starting` worker whose terminal it cannot find. + d.reconcileMissingWorkerTerminal(started.dispatch.id, 'runtime restarted') + + expect(() => + d.recordCreatedWorkerTerminalCustody({ + dispatchId: started.dispatch.id, + handle: 'term_worker', + paneKey: 'tab_w:leaf_w', + processIncarnation: 'pty_w:1', + worktreeId: 'repo::worktree' + }) + ).toThrow(/is not starting/) + expect(d.getWorkerTerminalResourceByOwner(started.dispatch.id)).toBeUndefined() + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-topology.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-topology.ts index d7c526696a9..b18eb9ee6c2 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-topology.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-topology.ts @@ -1,4 +1,5 @@ import type { AgentLaunchPreferences } from '../../../../../../shared/agent-session-host-authority' +import { narrowStructuredLaunchSeedOptions } from '../../../../../../shared/native-chat-session-option-defaults' import type { TuiAgent } from '../../../../../../shared/tui-agent' import type { OrcaRuntimeService } from '../../../../orca-runtime' import type { OrchestrationDb } from '../../../../orchestration/db' @@ -96,6 +97,8 @@ export async function createStructuredWorkerSessionForWorktree(args: { worktreeId: string agent: TuiAgent dispatchId: string + /** `--model`/`--effort`; the session seeds them exactly as a saved selection is seeded. */ + launchPreferences?: AgentLaunchPreferences effects: WorkerEffect[] }): Promise>> { if (args.agent !== 'claude' && args.agent !== 'codex') { @@ -104,11 +107,13 @@ export async function createStructuredWorkerSessionForWorktree(args: { `Structured workers support claude and codex; ${args.agent} has no structured session.` ) } + const options = narrowStructuredLaunchSeedOptions(args.launchPreferences) const created = await createStructuredWorkerSession({ runtime: args.runtime, worktreeId: args.worktreeId, agent: args.agent, dispatchId: args.dispatchId, + ...(options ? { options } : {}), onJournalActivity: (sessionId) => args.runtime.notifyStructuredSessionJournalActivity?.(sessionId) }) @@ -143,118 +148,6 @@ export function applyWaitForSetupOutcome( } } -export async function createWorkerWorktree(args: { - runtime: OrcaRuntimeService - db: OrchestrationDb - dispatchId: string - requestedWorktree: string - coordinatorWorktree: Awaited> - params: { - repo?: string - name?: string - baseBranch?: string - displayName?: string - comment?: string - setup?: 'run' | 'skip' | 'inherit' - from: string - } - agent: TuiAgent - launchPreferences?: AgentLaunchPreferences - effects: WorkerEffect[] -}): Promise<{ - worktree: Awaited> - terminalHandle: string - setupReceipt: WorkerSetupReceipt -}> { - const { runtime, db, dispatchId, requestedWorktree, coordinatorWorktree, params, effects } = args - const setupDecision = params.setup ?? 'run' - db.recordWorkerStage({ dispatchId, stage: 'worktree_creating', effects }) - const created = await runtime.createManagedWorktree({ - repoSelector: params.repo ?? coordinatorWorktree.repoId, - name: params.name as string, - baseBranch: params.baseBranch, - displayName: params.displayName, - ...(params.displayName !== undefined ? { displayNameKind: 'user' as const } : {}), - comment: params.comment, - // setupDecision runs setup without the legacy runHooks activation side effect. - runHooks: false, - setupDecision, - awaitTerminalProvisioning: true, - observeSetupCompletion: true, - createdWithAgent: args.agent, - startupAgent: args.agent, - ...(args.launchPreferences ? { startupLaunchPreferences: args.launchPreferences } : {}), - activate: false, - lineage: { - parentWorktree: requestedWorktree === 'new-child' ? coordinatorWorktree.id : undefined, - noParent: requestedWorktree === 'new-top-level', - callerTerminalHandle: params.from - } - }) - const terminalHandle = created.startupTerminal?.handle - effects.push({ - kind: 'worktree', - action: requestedWorktree === 'new-child' ? 'created_child' : 'created_top_level', - id: created.worktree.id - }) - db.recordWorkerStage({ - dispatchId, - stage: 'worktree_created', - worktreeId: created.worktree.id, - effects, - residualResources: effects - }) - const setupReceipt = { - requested: setupDecision, - effective: setupDecision, - source: params.setup ? 'explicit_request' : 'orchestration_default', - hookFound: created.setupReceipt?.hookFound ?? false, - startupPolicy: created.setupReceipt?.startupPolicy ?? 'start-immediately', - state: created.setupReceipt?.state ?? 'not_configured' - } - if (!terminalHandle) { - throw new Error(created.warning ?? 'Agent-first worktree creation returned no terminal.') - } - const listed = await runtime.listTerminals(`id:${created.worktree.id}`, undefined, { - includeVisualLayouts: false - }) - const setupTerminalHandle = created.setupReceipt?.terminalHandle - for (const terminal of listed.terminals) { - effects.push({ - kind: 'terminal', - role: - terminal.handle === terminalHandle - ? 'agent' - : terminal.handle === setupTerminalHandle - ? 'setup' - : 'configured_tab', - action: terminal.handle === terminalHandle ? 'reused_agent_terminal' : 'created', - id: terminal.handle, - tabId: terminal.tabId, - leafId: terminal.leafId - }) - } - const setupTerminal = effects.find( - (effect) => effect.kind === 'terminal' && effect.role === 'setup' - ) - effects.push({ - kind: 'setup', - action: setupDecision, - requested: setupReceipt.requested, - effective: setupReceipt.effective, - source: setupReceipt.source, - hookFound: setupReceipt.hookFound, - startupPolicy: setupReceipt.startupPolicy, - state: setupReceipt.state, - terminalId: setupTerminalHandle ?? setupTerminal?.id - }) - return { - worktree: created.worktree as Awaited>, - terminalHandle, - setupReceipt - } -} - export function monitorWorkerSetup(args: { runtime: OrcaRuntimeService db: OrchestrationDb diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-worktree-creation.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-worktree-creation.ts new file mode 100644 index 00000000000..fe0c11e374d --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-worktree-creation.ts @@ -0,0 +1,139 @@ +/** + * Creating the worktree a `--worktree new-child` / `new-top-level` dispatch asks for. + * + * `withAgentTerminal` is the whole difference between the two worker modes: a PTY worker's + * worktree is created agent-first, so the startup terminal IS the worker, while a structured + * worker's worktree is created with no agent at all and its session is created for the worktree + * afterwards. Setup, default tabs and lineage are identical either way. + */ + +import type { AgentLaunchPreferences } from '../../../../../../shared/agent-session-host-authority' +import type { TuiAgent } from '../../../../../../shared/tui-agent' +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import type { OrchestrationDb } from '../../../../orchestration/db' +import type { WorkerEffect, WorkerSetupReceipt } from './worker-topology' + +export async function createWorkerWorktree(args: { + runtime: OrcaRuntimeService + db: OrchestrationDb + dispatchId: string + requestedWorktree: string + coordinatorWorktree: Awaited> + params: { + repo?: string + name?: string + baseBranch?: string + displayName?: string + comment?: string + setup?: 'run' | 'skip' | 'inherit' + from: string + } + agent: TuiAgent + /** False for a structured worker: the session is created for the worktree afterwards, so the + * worktree must not be created agent-first. Setup and default tabs still run; what is skipped + * is the startup agent terminal and, with it, the TUI trust write — which a structured session + * does not need, since the trust preset exists so a PTY agent's menu does not eat the first + * bracketed paste. The renderer's own structured worktree create skips both the same way. */ + withAgentTerminal: boolean + launchPreferences?: AgentLaunchPreferences + effects: WorkerEffect[] +}): Promise<{ + worktree: Awaited> + terminalHandle: string | undefined + setupReceipt: WorkerSetupReceipt +}> { + const { runtime, db, dispatchId, requestedWorktree, coordinatorWorktree, params, effects } = args + const setupDecision = params.setup ?? 'run' + db.recordWorkerStage({ dispatchId, stage: 'worktree_creating', effects }) + const created = await runtime.createManagedWorktree({ + repoSelector: params.repo ?? coordinatorWorktree.repoId, + name: params.name as string, + baseBranch: params.baseBranch, + displayName: params.displayName, + ...(params.displayName !== undefined ? { displayNameKind: 'user' as const } : {}), + comment: params.comment, + // setupDecision runs setup without the legacy runHooks activation side effect. + runHooks: false, + setupDecision, + awaitTerminalProvisioning: true, + observeSetupCompletion: true, + createdWithAgent: args.agent, + ...(args.withAgentTerminal + ? { + startupAgent: args.agent, + ...(args.launchPreferences ? { startupLaunchPreferences: args.launchPreferences } : {}) + } + : {}), + activate: false, + lineage: { + parentWorktree: requestedWorktree === 'new-child' ? coordinatorWorktree.id : undefined, + noParent: requestedWorktree === 'new-top-level', + callerTerminalHandle: params.from + } + }) + const terminalHandle = created.startupTerminal?.handle + effects.push({ + kind: 'worktree', + action: requestedWorktree === 'new-child' ? 'created_child' : 'created_top_level', + id: created.worktree.id + }) + db.recordWorkerStage({ + dispatchId, + stage: 'worktree_created', + worktreeId: created.worktree.id, + effects, + residualResources: effects + }) + const setupReceipt = { + requested: setupDecision, + effective: setupDecision, + source: params.setup ? 'explicit_request' : 'orchestration_default', + hookFound: created.setupReceipt?.hookFound ?? false, + startupPolicy: created.setupReceipt?.startupPolicy ?? 'start-immediately', + state: created.setupReceipt?.state ?? 'not_configured' + } + if (args.withAgentTerminal && !terminalHandle) { + throw new Error(created.warning ?? 'Agent-first worktree creation returned no terminal.') + } + const listed = await runtime.listTerminals(`id:${created.worktree.id}`, undefined, { + includeVisualLayouts: false + }) + const setupTerminalHandle = created.setupReceipt?.terminalHandle + for (const terminal of listed.terminals) { + effects.push({ + kind: 'terminal', + role: + terminal.handle === terminalHandle + ? 'agent' + : terminal.handle === setupTerminalHandle + ? 'setup' + : 'configured_tab', + // Every terminal listed here — the agent terminal included — was created by this call's + // agent-first worktree creation. The old 'reused_agent_terminal' verb on the agent row + // conflated the role test with a lifecycle claim and misdiagnosed at least one incident. + action: 'created', + id: terminal.handle, + tabId: terminal.tabId, + leafId: terminal.leafId + }) + } + const setupTerminal = effects.find( + (effect) => effect.kind === 'terminal' && effect.role === 'setup' + ) + effects.push({ + kind: 'setup', + action: setupDecision, + requested: setupReceipt.requested, + effective: setupReceipt.effective, + source: setupReceipt.source, + hookFound: setupReceipt.hookFound, + startupPolicy: setupReceipt.startupPolicy, + state: setupReceipt.state, + terminalId: setupTerminalHandle ?? setupTerminal?.id + }) + return { + worktree: created.worktree as Awaited>, + terminalHandle, + setupReceipt + } +} diff --git a/src/main/runtime/rpc/methods/orchestration/worker/workers-new-worktree.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/workers-new-worktree.test.ts index a0d74031b72..5164ac0b22f 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/workers-new-worktree.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/workers-new-worktree.test.ts @@ -157,7 +157,7 @@ describe('orchestration new-worktree workers', () => { expect.objectContaining({ kind: 'terminal', role: 'agent', - action: 'reused_agent_terminal', + action: 'created', id: 'term_worker' }) ]) diff --git a/src/main/runtime/rpc/methods/session-tabs-inventory.ts b/src/main/runtime/rpc/methods/session-tabs-inventory.ts index aa3777ca12a..24a0af087ce 100644 --- a/src/main/runtime/rpc/methods/session-tabs-inventory.ts +++ b/src/main/runtime/rpc/methods/session-tabs-inventory.ts @@ -4,6 +4,7 @@ import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime- import type { RpcContext } from '../core' import { projectSessionTabAgentStatus } from './session-tab-agent-status-projection' import { projectSessionTabBrowserPlacements } from './session-tab-browser-placement-projection' +import { createSessionTabsRetirementProofDelta } from './session-tabs-retirement-proof-delta' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' type SessionTabsInventory = { @@ -117,6 +118,7 @@ export async function subscribeSessionTabsInventory( const deliveredChangeSequenceByWorktree = new Map() let censusChangeSequence: number | undefined let censusInvalidated = false + const withProofDelta = createSessionTabsRetirementProofDelta(context.clientCapabilities) const projectChange = (snapshot: SessionTabsChange): SessionTabsChange => projectSessionTabsForClient( snapshot, @@ -193,7 +195,7 @@ export async function subscribeSessionTabsInventory( } emit({ type: 'updated', - ...projected + ...withProofDelta(projected) }) if (projected.removed === true) { publishedSnapshotsByWorktree.delete(snapshot.worktree) @@ -267,7 +269,7 @@ export async function subscribeSessionTabsInventory( } const { inventory, changeSequence } = collected censusChangeSequence = changeSequence - emit({ type: 'snapshots', ...inventory }) + emit({ type: 'snapshots', ...inventory, snapshots: inventory.snapshots.map(withProofDelta) }) for (const snapshot of inventory.snapshots) { publishedSnapshotsByWorktree.set(snapshot.worktree, withoutNavigationIntent(snapshot)) } diff --git a/src/main/runtime/rpc/methods/session-tabs-retirement-proof-delta.test.ts b/src/main/runtime/rpc/methods/session-tabs-retirement-proof-delta.test.ts new file mode 100644 index 00000000000..021b6f20622 --- /dev/null +++ b/src/main/runtime/rpc/methods/session-tabs-retirement-proof-delta.test.ts @@ -0,0 +1,158 @@ +import { describe, expect, it, vi } from 'vitest' +import type { OrcaRuntimeService } from '../../orca-runtime' +import { SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { + RuntimeMobileSessionRetiredTerminalSurface, + RuntimeMobileSessionTabsResult +} from '../../../../shared/runtime-types' +import { RpcDispatcher } from '../dispatcher' +import { SESSION_TAB_METHODS } from './session-tabs' +import { createSessionTabsRetirementProofDelta } from './session-tabs-retirement-proof-delta' + +const WORKTREE = 'wt-proofs' + +function proof(index: number): RuntimeMobileSessionRetiredTerminalSurface { + return { + parentTabId: `tab-${index}`, + leafId: `leaf-${index}`, + ptyId: `pty-${index}`, + terminal: `term_${index}`, + incarnationId: `inc-${index}` + } +} + +function frame( + snapshotVersion: number, + retiredTerminalSurfaces?: RuntimeMobileSessionRetiredTerminalSurface[] +): RuntimeMobileSessionTabsResult { + return { + worktree: WORKTREE, + publicationEpoch: 'epoch', + snapshotVersion, + activeGroupId: null, + activeTabId: null, + activeTabType: null, + ...(retiredTerminalSurfaces ? { retiredTerminalSurfaces } : {}), + tabs: [] + } +} + +describe('session tabs retirement proof delta', () => { + it('passes every frame through untouched for a client that did not negotiate it', () => { + const project = createSessionTabsRetirementProofDelta(undefined) + const full = frame(1, [proof(1), proof(2)]) + expect(project(full)).toBe(full) + expect(project(frame(2, [proof(1), proof(2)]))).toEqual(frame(2, [proof(1), proof(2)])) + }) + + // Why `[]` rather than omitting the field: absence is the host's "I hold no proofs" signal and + // tells the client to forget, so a delta with nothing new must stay distinguishable from it. + it('sends each proof once and an empty list when nothing is new', () => { + const project = createSessionTabsRetirementProofDelta([ + SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY + ]) + expect(project(frame(1, [proof(1)]))).toEqual(frame(1, [proof(1)])) + expect(project(frame(2, [proof(1)]))).toEqual(frame(2, [])) + expect(project(frame(3, [proof(1), proof(2)]))).toEqual(frame(3, [proof(2)])) + expect(project(frame(4, [proof(1), proof(2)]))).toEqual(frame(4, [])) + }) + + it('resends a proof that left the host list and came back', () => { + const project = createSessionTabsRetirementProofDelta([ + SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY + ]) + project(frame(1, [proof(1)])) + // Surface revived: the host dropped the proof and publishes an empty list. + expect(project(frame(2, []))).toEqual(frame(2, [])) + expect(project(frame(3, [proof(1)]))).toEqual(frame(3, [proof(1)])) + }) + + it('starts over for a worktree after a removed frame or a proof-less host', () => { + const project = createSessionTabsRetirementProofDelta([ + SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY + ]) + project(frame(1, [proof(1)])) + project({ ...frame(2), removed: true } as RuntimeMobileSessionTabsResult) + expect(project(frame(3, [proof(1)]))).toEqual(frame(3, [proof(1)])) + project(frame(4)) + expect(project(frame(5, [proof(1)]))).toEqual(frame(5, [proof(1)])) + }) +}) + +// Why: the host pins up to 64 proofs per worktree for its lifetime, so this is the steady-state +// cost of every title tick on a churn-heavy worktree for a paired mobile/relay/SSH client. +describe('session.tabs.subscribe retirement proof payload', () => { + // Real identities are UUID-sized: tab/leaf/pty ids and `term_` handles. + const uuid = (index: number): string => + `${index.toString(16).padStart(8, '0')}-4a1b-4c2d-8e3f-000000000000` + const proofs = Array.from({ length: 64 }, (_, index) => ({ + parentTabId: `terminal-${uuid(index)}`, + leafId: uuid(index + 1000), + ptyId: uuid(index + 2000), + terminal: `term_${uuid(index + 3000)}`, + incarnationId: uuid(index + 4000) + })) + + async function subscribeAndTick(clientCapabilities: readonly string[] | undefined): Promise<{ + initial: string + tick: string + }> { + let listener: ((snapshot: RuntimeMobileSessionTabsResult) => void) | undefined + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: () => ({}), + listMobileSessionTabs: vi.fn().mockResolvedValue(frame(1, proofs)), + registerSubscriptionCleanup: vi.fn(), + onMobileSessionTabsChanged: vi.fn( + (next: (snapshot: RuntimeMobileSessionTabsResult) => void) => { + listener = next + return () => {} + } + ) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + const messages: string[] = [] + await dispatcher.dispatchStreaming( + { + id: 'req-1', + authToken: 'tok', + method: 'session.tabs.subscribe', + params: { worktree: 'id:wt' } + }, + (message) => messages.push(message), + { clientKind: 'runtime', clientCapabilities } + ) + // An OSC title change bumps the version and republishes the same 64 proofs. + listener!(frame(2, proofs)) + return { initial: messages[0]!, tick: messages[1]! } + } + + // Why: a reconnect is a new subscribe with fresh per-stream state, and the client's ledger + // resets on its new connection generation — so the first frame must carry the full set. + it('resends the full proof set on the first frame of a fresh stream', async () => { + const first = await subscribeAndTick([SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY]) + const reconnected = await subscribeAndTick([ + SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY + ]) + expect(JSON.parse(first.initial).result.retiredTerminalSurfaces).toEqual(proofs) + expect(JSON.parse(reconnected.initial).result.retiredTerminalSurfaces).toEqual(proofs) + }) + + it('drops the repeated proof list from a title tick for a negotiated client', async () => { + const legacy = await subscribeAndTick(undefined) + const delta = await subscribeAndTick([SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY]) + + const legacyTick = JSON.parse(legacy.tick).result + const deltaTick = JSON.parse(delta.tick).result + expect(legacyTick.retiredTerminalSurfaces).toHaveLength(64) + expect(deltaTick.retiredTerminalSurfaces).toEqual([]) + // Both clients still receive the full list on the initial snapshot. + expect(JSON.parse(legacy.initial).result.retiredTerminalSurfaces).toHaveLength(64) + expect(JSON.parse(delta.initial).result.retiredTerminalSurfaces).toHaveLength(64) + + // The delta tick keeps a two-byte `[]` so the client can tell "nothing new" from "no proofs". + const proofBytes = Buffer.byteLength(JSON.stringify(proofs)) + expect(proofBytes).toBeGreaterThan(8_000) + expect(Buffer.byteLength(legacy.tick) - Buffer.byteLength(delta.tick)).toBe(proofBytes - 2) + }) +}) diff --git a/src/main/runtime/rpc/methods/session-tabs-retirement-proof-delta.ts b/src/main/runtime/rpc/methods/session-tabs-retirement-proof-delta.ts new file mode 100644 index 00000000000..059af124a78 --- /dev/null +++ b/src/main/runtime/rpc/methods/session-tabs-retirement-proof-delta.ts @@ -0,0 +1,45 @@ +import { + SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY, + type RuntimeCapability +} from '../../../../shared/protocol-version' +import type { RuntimeMobileSessionRetiredTerminalSurface } from '../../../../shared/runtime-types' +import { retirementProofKey as proofKey } from '../../../../shared/terminal-retirement-proof-ledger' + +type ProofCarrier = { + worktree: string + removed?: true + retiredTerminalSurfaces?: RuntimeMobileSessionRetiredTerminalSurface[] +} + +export type SessionTabsRetirementProofDelta = (frame: TFrame) => TFrame + +/** + * Per-stream projection that sends each retirement proof once. The host pins up to 64 proofs per + * worktree for its process lifetime, so without this every title tick re-ships the whole list. + * A capable client retains what it was sent; a legacy client keeps receiving the full list. + */ +export function createSessionTabsRetirementProofDelta( + clientCapabilities: readonly RuntimeCapability[] | undefined +): SessionTabsRetirementProofDelta { + if (!clientCapabilities?.includes(SESSION_TABS_RETIREMENT_PROOF_DELTA_RUNTIME_CAPABILITY)) { + return (frame) => frame + } + const sentByWorktree = new Map>() + return (frame: TFrame): TFrame => { + if (frame.removed === true || frame.retiredTerminalSurfaces === undefined) { + sentByWorktree.delete(frame.worktree) + return frame + } + const sent = sentByWorktree.get(frame.worktree) + const fresh = sent + ? frame.retiredTerminalSurfaces.filter((proof) => !sent.has(proofKey(proof))) + : frame.retiredTerminalSurfaces + // Why: track exactly the current list, so a proof that leaves and returns is sent again. + sentByWorktree.set(frame.worktree, new Set(frame.retiredTerminalSurfaces.map(proofKey))) + // Why: an empty list is a real signal ("nothing new, keep yours"). Omitting the field would be + // indistinguishable from a host that holds no proofs, which is what tells the client to forget. + return fresh.length === frame.retiredTerminalSurfaces.length + ? frame + : { ...frame, retiredTerminalSurfaces: fresh } + } +} diff --git a/src/main/runtime/rpc/methods/session-tabs.ts b/src/main/runtime/rpc/methods/session-tabs.ts index 6296c462a39..aa0e0b24939 100644 --- a/src/main/runtime/rpc/methods/session-tabs.ts +++ b/src/main/runtime/rpc/methods/session-tabs.ts @@ -14,6 +14,7 @@ import { } from './session-tabs-inventory' import { SESSION_TAB_MARKDOWN_METHODS } from './session-tab-markdown-methods' import { SESSION_TAB_MUTATION_METHODS } from './session-tab-mutation-methods' +import { createSessionTabsRetirementProofDelta } from './session-tabs-retirement-proof-delta' import { restoreStructuredTabsIfSupported } from './structured-session-tab-restore' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' import { assertLegacyAiVaultResumeCommandAllowed } from '../../../ai-vault/structured-session-ownership' @@ -115,13 +116,16 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ if (closed) { return } + const withProofDelta = createSessionTabsRetirementProofDelta(clientCapabilities) emit({ type: 'snapshot', - ...projectSessionTabsForClient( - initial, - clientKind, - clientCapabilities, - isStructuredNativeChatEnabled(runtime) + ...withProofDelta( + projectSessionTabsForClient( + initial, + clientKind, + clientCapabilities, + isStructuredNativeChatEnabled(runtime) + ) ) }) initialized = true @@ -133,11 +137,13 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ if (snapshot.worktree === subscribedWorktree) { emit({ type: 'updated', - ...projectSessionTabsForClient( - snapshot, - clientKind, - clientCapabilities, - isStructuredNativeChatEnabled(runtime) + ...withProofDelta( + projectSessionTabsForClient( + snapshot, + clientKind, + clientCapabilities, + isStructuredNativeChatEnabled(runtime) + ) ) }) } diff --git a/src/main/runtime/rpc/methods/settled-worker-resume-fence-sweep.ts b/src/main/runtime/rpc/methods/settled-worker-resume-fence-sweep.ts deleted file mode 100644 index e3aac0e5803..00000000000 --- a/src/main/runtime/rpc/methods/settled-worker-resume-fence-sweep.ts +++ /dev/null @@ -1,45 +0,0 @@ -import type { OrcaRuntimeService } from '../../orca-runtime' -import type { RpcMethod } from '../core' - -/** - * One pass both stamps the automatic-resume fence on every settled worker pane and lifts it from - * every pane the recovery plan no longer claims. A fenced pane refuses a fresh spawn, so any path - * that drops a worker's row from that plan — release, user retain, user takeover — has to run the - * sweep in the same call, or the fence outlives its dispatch and the pane stays unspawnable until - * the next app start. Failures are swallowed: a fence sweep must never fail the RPC behind it. - */ -export function sweepSettledWorkerResumeFences(runtime: OrcaRuntimeService): void { - try { - runtime.prepareLegacyWorkerTerminalRecovery() - } catch (error) { - console.warn('[orchestration] settled worker resume fence sweep failed', error) - } -} - -/** Settling a worker is what makes its pane fenceable, and release/retain/takeover are what make it - * unfenceable again — so every one of those has to sweep in the same call. Without the settlement - * half the fence only appeared at the next app start, and reopening the pane in the same session - * respawned the agent. */ -const FENCE_SWEEPING_METHOD_NAMES = new Set([ - 'orchestration.workerRelease', - 'orchestration.workerRetain', - 'orchestration.workerStop', - 'orchestration.workerAbandon', - // Reusing a settled worker's pane for a new Dispatch drops the old row from the plan; without - // this the stale fence stays on the pane it just relaunched into. - 'orchestration.workerStart' -]) - -export function sweepingSettledWorkerResumeFences(method: RpcMethod): RpcMethod { - if (!FENCE_SWEEPING_METHOD_NAMES.has(method.name)) { - return method - } - return { - ...method, - handler: async (params, ctx) => { - const result = await method.handler(params, ctx) - sweepSettledWorkerResumeFences(ctx.runtime) - return result - } - } -} diff --git a/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts new file mode 100644 index 00000000000..cc5a7282080 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts @@ -0,0 +1,105 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionBackgroundTaskState } from '../../../../shared/agent-session-wire' +import { AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY } from '../../../../shared/protocol-version' +import { remoteRuntimeClientCapabilities } from '../../../../shared/remote-runtime-client-capabilities' +import type { AgentSessionSubscribeInput } from '../../../native-chat/agent-session-wire/structured-agent-session-subscribers' +import { + call, + clearStructuredHostStub, + hostCalls, + installStructuredHostStub, + SESSION, + STRUCTURED_CLIENT +} from './structured-agent-session-rpc.test-fixture' + +beforeEach(installStructuredHostStub) +afterEach(clearStructuredHostStub) + +const TASKS: AgentSessionBackgroundTaskState = { + state: 'monitoring', + supportsStopAll: false, + tasks: [{ id: 'child', kind: 'agent' }] +} +const CURRENT_CLIENT = { + ...STRUCTURED_CLIENT, + clientCapabilities: remoteRuntimeClientCapabilities(STRUCTURED_CLIENT.clientCapabilities) +} + +describe('background-task stop capability at the RPC boundary', () => { + it('advertises reader support on remote requests and subscriptions', () => { + expect(CURRENT_CLIENT.clientCapabilities).toContain( + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY + ) + }) + + it.each([ + ['legacy reader', STRUCTURED_CLIENT, null], + ['current reader', CURRENT_CLIENT, TASKS], + ['in-process reader', undefined, TASKS] + ] as const)('projects history for a %s', async (_label, client, expected) => { + hostCalls.history.mockReturnValue({ ok: true, page: { items: [], backgroundTasks: TASKS } }) + expect( + await call('agentSession.history', { sessionId: SESSION, direction: 'tail' }, client) + ).toMatchObject({ ok: true, result: { page: { backgroundTasks: expected } } }) + }) + + it.each(['snapshot', 'batch', 'reset'] as const)( + 'gates the %s stream without changing the provider state', + async (type) => { + hostCalls.hold = vi.fn(async () => undefined) + hostCalls.subscribe.mockImplementation((input: AgentSessionSubscribeInput) => { + const base = { sessionId: SESSION, fence: 1, backgroundTasks: TASKS } + if (type === 'batch') { + input.emit({ + ...base, + type, + batch: { + cursor: { epoch: 'a', sequence: 0 }, + items: [], + removedItemIds: [], + submissions: [] + } + }) + } else { + const page = { + sessionId: SESSION, + epoch: 'a', + direction: 'tail' as const, + items: [], + removedItemIds: [], + submissions: [], + window: { oldest: null, newest: null, nextCursor: { epoch: 'a', sequence: 0 } }, + hasOlder: false, + hasNewer: false + } + input.emit( + type === 'snapshot' + ? { ...base, type, page } + : { ...base, type, page, reset: 'epoch_changed' } + ) + } + return () => {} + }) + for (const [client, expected] of [ + [STRUCTURED_CLIENT, null], + [CURRENT_CLIENT, TASKS] + ] as const) { + expect(await call('agentSession.subscribe', { sessionId: SESSION }, client)).toMatchObject({ + ok: true, + result: { type, backgroundTasks: expected } + }) + } + expect(TASKS.supportsStopAll).toBe(false) + } + ) + + it('preserves legacy stoppable state for both readers', async () => { + const stoppable = { state: 'monitoring', tasks: TASKS.tasks } + hostCalls.history.mockReturnValue({ ok: true, page: { items: [], backgroundTasks: stoppable } }) + for (const client of [STRUCTURED_CLIENT, CURRENT_CLIENT]) { + expect( + await call('agentSession.history', { sessionId: SESSION, direction: 'tail' }, client) + ).toMatchObject({ ok: true, result: { page: { backgroundTasks: stoppable } } }) + } + }) +}) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts new file mode 100644 index 00000000000..02afc5569bf --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts @@ -0,0 +1,47 @@ +import type { + AgentSessionBackgroundTaskState, + AgentSessionHistoryResult, + AgentSessionSubscribeEvent +} from '../../../../shared/agent-session-wire' +import { AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY } from '../../../../shared/protocol-version' +import type { RpcContext } from '../core' + +type BackgroundTaskReader = Pick + +function supportsReadOnlyTasks(ctx: BackgroundTaskReader): boolean { + return ( + ctx.clientKind === undefined || + ctx.clientCapabilities?.includes(AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY) === true + ) +} + +function projectState( + state: AgentSessionBackgroundTaskState | null | undefined, + ctx: BackgroundTaskReader +): AgentSessionBackgroundTaskState | null | undefined { + // Legacy readers always offer a stop; retain their pre-producer empty strip. + return state?.supportsStopAll === false && !state.supportsTaskStop && !supportsReadOnlyTasks(ctx) + ? null + : state +} + +export function projectBackgroundTaskHistory( + result: AgentSessionHistoryResult, + ctx: BackgroundTaskReader +): AgentSessionHistoryResult { + const state = projectState(result.page.backgroundTasks, ctx) + return state === result.page.backgroundTasks + ? result + : { ...result, page: { ...result.page, backgroundTasks: state } } +} + +export function projectBackgroundTaskEvent( + event: AgentSessionSubscribeEvent, + ctx: BackgroundTaskReader +): AgentSessionSubscribeEvent { + if (!('backgroundTasks' in event)) { + return event + } + const state = projectState(event.backgroundTasks, ctx) + return state === event.backgroundTasks ? event : { ...event, backgroundTasks: state } +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session-create.ts b/src/main/runtime/rpc/methods/structured-agent-session-create.ts index b0ce4666861..cad5671765e 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-create.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-create.ts @@ -49,6 +49,10 @@ export async function prepareStructuredAgentSessionCreateForWorktree(args: { agent: 'claude' | 'codex' caller: StructuredAgentSessionCaller resumeFrom?: StructuredAgentSessionResumeSource + /** Replaces the seed options the host resolves from settings. Orchestration passes the + * `--model`/`--effort` the dispatch asked for; a chat the user opened passes nothing and keeps + * the saved selection. Narrowed by the caller, so `{}` never reaches the reservation. */ + options?: Readonly> }): Promise { // Adoption replay may need the record loaded from disk before source discovery can be skipped. let host = args.resumeFrom ? await args.ensureHost() : null @@ -70,6 +74,10 @@ export async function prepareStructuredAgentSessionCreateForWorktree(args: { host, attachParams: { ...resolvedAttach, + // After the fingerprint, deliberately: `attachFingerprintFields` excludes options because + // they are the session's initial state, not its identity, so a retry that re-resolves them + // must replay rather than conflict. + ...(args.options ? { options: args.options } : {}), provider: resolved.provider as 'claude' | 'codex', agent: resolved.agent as 'claude' | 'codex', envelope: { ...args.envelope, payloadFingerprint: hostFingerprint } @@ -121,6 +129,7 @@ export async function createStructuredAgentSessionForWorktree(args: { worktree: string agent: 'claude' | 'codex' activate: boolean + options?: Readonly> }): Promise> { const prepared: PreparedStructuredAgentSessionCreate | StructuredCreateRefused = await resolveUncommittedStructuredCreate(() => diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index 3078c9fff61..d629b1c29a1 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -10,6 +10,10 @@ import { computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' import type { z } from 'zod' +import { + projectBackgroundTaskEvent, + projectBackgroundTaskHistory +} from './structured-agent-session-background-task-capability' import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled as ensureHostInstalled, @@ -251,7 +255,8 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ defineMethod({ name: 'agentSession.history', params: HistoryParams, - handler: async (params, ctx) => requireHost(ctx).history(params) + handler: async (params, ctx) => + projectBackgroundTaskHistory(requireHost(ctx).history(params), ctx) }), defineStreamingMethod({ name: 'agentSession.subscribe', @@ -278,7 +283,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ dispose = host.subscribe({ id: subscriptionId, sessionId: params.sessionId, - emit, + emit: (event) => emit(projectBackgroundTaskEvent(event, ctx)), ...(params.cursor ? { cursor: params.cursor } : {}) }) if (stream.isClosed()) { diff --git a/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts b/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts index 82cc53ff735..151285ffb1e 100644 --- a/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts +++ b/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts @@ -62,7 +62,10 @@ describe('worker-stop on a structured worker this runtime cannot reach', () => { worktreeId: WORKTREE, hostScope: { kind: 'local', hostId: 'local' } }) - const task = db.createTask({ spec: 'stop a structured worker' }) + const task = db.createTask({ + runId: 'run_legacy_local', + spec: 'stop a structured worker' + }) const started = db.createStartingWorkerDispatch({ creator: { kind: 'system' }, maxDepth: Number.MAX_SAFE_INTEGER, diff --git a/src/main/runtime/rpc/orchestration-11745-regression-verification.test.ts b/src/main/runtime/rpc/orchestration-11745-regression-verification.test.ts index 47d585ca244..7a644cc9def 100644 --- a/src/main/runtime/rpc/orchestration-11745-regression-verification.test.ts +++ b/src/main/runtime/rpc/orchestration-11745-regression-verification.test.ts @@ -642,7 +642,11 @@ function createAdoptedDb(options: { settleWork: boolean }): { const dbPath = join(dir, 'orchestration.db') const before = new OrchestrationDb(dbPath) - const task = before.createTask({ spec: 'legacy assignment', createdByTerminalHandle: 'term_old' }) + const task = before.createTask({ + runId: 'run_legacy_local', + spec: 'legacy assignment', + createdByTerminalHandle: 'term_old' + }) createRootDispatch( before, task.id, @@ -650,6 +654,7 @@ function createAdoptedDb(options: { settleWork: boolean }): { 'tab_old:33333333-3333-4333-8333-333333333333' ) const recovery = before.insertMessage({ + runId: 'run_legacy_local', from: 'term_old_worker', to: 'term_old', subject: 'recovered worker outcome', diff --git a/src/main/runtime/rpc/orchestration-legacy-compatibility-dispatcher-test-fixture.ts b/src/main/runtime/rpc/orchestration-legacy-compatibility-dispatcher-test-fixture.ts index cca68da8f63..faf934a6d66 100644 --- a/src/main/runtime/rpc/orchestration-legacy-compatibility-dispatcher-test-fixture.ts +++ b/src/main/runtime/rpc/orchestration-legacy-compatibility-dispatcher-test-fixture.ts @@ -53,6 +53,7 @@ export function createHarness(): LegacyCompatibilityDispatcherHarness { const dbPath = join(dir, 'orchestration.db') const before = new OrchestrationDb(dbPath) const task = before.createTask({ + runId: 'run_legacy_local', spec: 'legacy assignment', createdByTerminalHandle: COORDINATOR_HANDLE }) diff --git a/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts b/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts index 3040c37a9ef..71daf09b0e3 100644 --- a/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts +++ b/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts @@ -43,6 +43,7 @@ function createHarness(): Harness { const dbPath = join(dir, 'orchestration.db') const before = new OrchestrationDb(dbPath) const task = before.createTask({ + runId: 'run_legacy_local', spec: 'legacy assignment', createdByTerminalHandle: COORDINATOR_HANDLE }) diff --git a/src/main/runtime/rpc/orchestration-legacy-question-takeover.test.ts b/src/main/runtime/rpc/orchestration-legacy-question-takeover.test.ts index 77d7e2d5a02..82c59f39587 100644 --- a/src/main/runtime/rpc/orchestration-legacy-question-takeover.test.ts +++ b/src/main/runtime/rpc/orchestration-legacy-question-takeover.test.ts @@ -40,12 +40,14 @@ function createHarness(options?: { seedCutoverQuestion?: boolean; seedCutoverAns const dbPath = join(dir, 'orchestration.db') const before = new OrchestrationDb(dbPath) const task = before.createTask({ + runId: 'run_legacy_local', spec: 'legacy assignment', createdByTerminalHandle: COORDINATOR_HANDLE }) const dispatch = createRootDispatch(before, task.id, WORKER_HANDLE, WORKER_PANE) const cutoverQuestion = options?.seedCutoverQuestion ? before.insertMessage({ + runId: 'run_legacy_local', from: WORKER_HANDLE, to: COORDINATOR_HANDLE, subject: 'Question', @@ -60,6 +62,7 @@ function createHarness(options?: { seedCutoverQuestion?: boolean; seedCutoverAns const cutoverAnswer = cutoverQuestion && options?.seedCutoverAnswer ? before.insertMessage({ + runId: 'run_legacy_local', from: COORDINATOR_HANDLE, to: WORKER_HANDLE, subject: 'Re: Question', @@ -308,7 +311,10 @@ describe('legacy question takeover compatibility', () => { resumed as { result: { legacyCompatibility: { - answerAcknowledgement: { questionId: string; answerMessageId: string } + answerAcknowledgement: { + questionId: string + answerMessageId: string + } } } } diff --git a/src/main/runtime/rpc/orchestration-legacy-takeover-delivery.test.ts b/src/main/runtime/rpc/orchestration-legacy-takeover-delivery.test.ts index bcaf5fca11f..feb3017b538 100644 --- a/src/main/runtime/rpc/orchestration-legacy-takeover-delivery.test.ts +++ b/src/main/runtime/rpc/orchestration-legacy-takeover-delivery.test.ts @@ -51,6 +51,7 @@ function createHarness(): Harness { const dbPath = join(dir, 'orchestration.db') const before = new OrchestrationDb(dbPath) const task = before.createTask({ + runId: 'run_legacy_local', spec: 'legacy assignment', createdByTerminalHandle: COORDINATOR_HANDLE }) diff --git a/src/main/runtime/rpc/orchestration-legacy-takeover-dispatcher.test.ts b/src/main/runtime/rpc/orchestration-legacy-takeover-dispatcher.test.ts index 2a2d6b4937b..e5a05fe7d26 100644 --- a/src/main/runtime/rpc/orchestration-legacy-takeover-dispatcher.test.ts +++ b/src/main/runtime/rpc/orchestration-legacy-takeover-dispatcher.test.ts @@ -47,6 +47,7 @@ function createHarness(): Harness { const dbPath = join(dir, 'orchestration.db') const before = new OrchestrationDb(dbPath) const task = before.createTask({ + runId: 'run_legacy_local', spec: 'legacy assignment', createdByTerminalHandle: COORDINATOR_HANDLE }) diff --git a/src/main/runtime/rpc/orchestration-mutation-ledger.test.ts b/src/main/runtime/rpc/orchestration-mutation-ledger.test.ts index d44d3f153d7..b87f595f4b2 100644 --- a/src/main/runtime/rpc/orchestration-mutation-ledger.test.ts +++ b/src/main/runtime/rpc/orchestration-mutation-ledger.test.ts @@ -44,7 +44,7 @@ describe('durable orchestration mutation ledger', () => { const runtime = new OrcaRuntimeService() runtime.setOrchestrationDb(db) const effect = vi.fn((subject: string) => - db.insertMessage({ from: 'caller', to: 'recipient', subject }) + db.insertMessage({ runId: 'run_legacy_local', from: 'caller', to: 'recipient', subject }) ) const dispatcher = new RpcDispatcher({ runtime, @@ -299,7 +299,10 @@ describe('durable orchestration mutation ledger', () => { const db = new OrchestrationDb(':memory:') const runtime = new OrcaRuntimeService() runtime.setOrchestrationDb(db) - const params = { from: 'term_coord', task: db.createTask({ spec: 'restart' }).id } + const params = { + from: 'term_coord', + task: db.createTask({ runId: 'run_legacy_local', spec: 'restart' }).id + } const callerFingerprint = db.getOrCreateLocalMutationCallerFingerprint() const payloadHash = createHash('sha256') .update(JSON.stringify({ method: 'orchestration.workerStart', params })) diff --git a/src/main/runtime/rpc/orchestration-mutation-request-show.test.ts b/src/main/runtime/rpc/orchestration-mutation-request-show.test.ts index cb31ea22736..e64fb5b9640 100644 --- a/src/main/runtime/rpc/orchestration-mutation-request-show.test.ts +++ b/src/main/runtime/rpc/orchestration-mutation-request-show.test.ts @@ -22,7 +22,7 @@ function createHarness() { const runtime = new OrcaRuntimeService() runtime.setOrchestrationDb(db) const effect = vi.fn((subject: string) => - db.insertMessage({ from: 'caller', to: 'recipient', subject }) + db.insertMessage({ runId: 'run_legacy_local', from: 'caller', to: 'recipient', subject }) ) const dispatcher = new RpcDispatcher({ runtime, diff --git a/src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts b/src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts index b2d3d110627..4172097853d 100644 --- a/src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts +++ b/src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts @@ -62,6 +62,7 @@ function createUpdateHarness(): Harness { const oldRuntimeDb = new OrchestrationDb(dbPath) const task = oldRuntimeDb.createTask({ + runId: 'run_legacy_local', spec: 'finish work across an app update', createdByTerminalHandle: COORDINATOR_HANDLE }) diff --git a/src/main/runtime/runtime-legacy-worker-terminal-recovery-controller.ts b/src/main/runtime/runtime-legacy-worker-terminal-recovery-controller.ts index cbb28e20336..ce034d8e349 100644 --- a/src/main/runtime/runtime-legacy-worker-terminal-recovery-controller.ts +++ b/src/main/runtime/runtime-legacy-worker-terminal-recovery-controller.ts @@ -22,10 +22,6 @@ export class RuntimeLegacyWorkerTerminalRecoveryController { constructor(private readonly ports: LegacyWorkerRecoveryPorts) {} - prepare(): LegacyWorkerTerminalRecoveryPlan { - return this.ports.preparePlan() - } - reconcile( options: LegacyWorkerRecoveryOptions = {} ): Promise { diff --git a/src/main/runtime/runtime-legacy-worker-terminal-recovery-persistence.ts b/src/main/runtime/runtime-legacy-worker-terminal-recovery-persistence.ts index 613718de7e5..df794567737 100644 --- a/src/main/runtime/runtime-legacy-worker-terminal-recovery-persistence.ts +++ b/src/main/runtime/runtime-legacy-worker-terminal-recovery-persistence.ts @@ -1,4 +1,4 @@ -import { LOCAL_EXECUTION_HOST_ID, type ExecutionHostId } from '../../shared/execution-host' +import type { ExecutionHostId } from '../../shared/execution-host' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { retireTerminalSurfaceFromPersistence } from './mobile-session-terminal-persistence-retirement' import type { OrchestrationDb } from './orchestration/db' @@ -18,144 +18,11 @@ export class RuntimeLegacyWorkerTerminalRecoveryPersistence { constructor( private readonly getStore: () => RuntimeStore | null, private readonly getDb: () => OrchestrationDb, - private readonly getHostId: (worktreeId: string) => ExecutionHostId | null, - /** The store write only reaches the next app start; a live renderer holds its own copy. */ - private readonly notifyFenceChanged?: (paneKey: string, blocked: boolean) => void + private readonly getHostId: (worktreeId: string) => ExecutionHostId | null ) {} - /** Panes announced as fenced before any sleeping record existed; the only place a lift for one - * can come from, because `liftRetiredFences` can only see panes that already have a record. */ - private readonly announcedBlockedPaneKeys = new Set() - prepare(): LegacyWorkerTerminalRecoveryPlan { - const plan = this.getPlan() - if (!plan) { - // An unreadable plan is not evidence that any pane stopped needing its fence: stamp - // nothing, lift nothing, retry on the next pass. - return { blockedPanes: [], candidates: [], ambiguousDispatchIds: [] } - } - const store = this.getStore() - if ( - !store?.getWorkspaceSession || - !store.setWorkspaceSession || - (!store.flushPendingOrThrowAsync && !store.flushOrThrow) - ) { - return plan - } - const sessions = new Map< - ExecutionHostId, - { current: WorkspaceSessionState; next: WorkspaceSessionState } - >() - const changedHostIds = new Set() - const fenceChanges: [string, boolean][] = [] - for (const blocked of plan.blockedPanes) { - // A worker can settle while its tab is still open, so there is no sleeping record to stamp - // yet. Tell the live renderer anyway: it mints the record on close and must fence it there. - if (!this.announcedBlockedPaneKeys.has(blocked.paneKey)) { - this.announcedBlockedPaneKeys.add(blocked.paneKey) - fenceChanges.push([blocked.paneKey, true]) - } - let hostIds: ExecutionHostId[] - try { - const hostId = this.getHostId(blocked.worktreeId) - if (!hostId) { - throw new Error('folder_workspace_not_found') - } - hostIds = [hostId] - } catch (error) { - console.warn('[orchestration] legacy worker resume fence owner is unavailable', { - worktreeId: blocked.worktreeId, - error - }) - hostIds = store.getWorkspaceSessionHostIds?.() ?? [LOCAL_EXECUTION_HOST_ID] - } - for (const hostId of hostIds) { - let state = sessions.get(hostId) - if (!state) { - const current = store.getWorkspaceSession(hostId) - if (!current) { - continue - } - state = { current, next: structuredClone(current) } - sessions.set(hostId, state) - } - const record = state.next.sleepingAgentSessionsByPaneKey?.[blocked.paneKey] - if ( - !record || - !runtimeWorktreeIdsEqual(record.worktreeId, blocked.worktreeId) || - record.automaticResumeBlockedBy === 'legacy-orchestration-worker' - ) { - continue - } - state.next.sleepingAgentSessionsByPaneKey = { - ...state.next.sleepingAgentSessionsByPaneKey, - [blocked.paneKey]: { ...record, automaticResumeBlockedBy: 'legacy-orchestration-worker' } - } - changedHostIds.add(hostId) - } - } - this.liftRetiredFences(store, plan, sessions, changedHostIds, fenceChanges) - const changed = [...sessions].filter(([hostId]) => changedHostIds.has(hostId)) - try { - for (const [hostId, state] of changed) { - store.setWorkspaceSession(state.next, hostId) - } - } catch (error) { - console.warn('[orchestration] failed to stage legacy worker resume fence', error) - return plan - } - for (const [paneKey, blocked] of fenceChanges) { - this.notifyFenceChanged?.(paneKey, blocked) - } - return plan - } - - /** A fence that outlives its dispatch leaves a pane that can never spawn again, so release, - * retain, user takeover and dispatch pruning — each of which drops the row from the plan — - * retire it here. An unreadable plan yields no blocked panes, so callers must not sweep. */ - private liftRetiredFences( - store: RuntimeStore, - plan: LegacyWorkerTerminalRecoveryPlan, - sessions: Map, - changedHostIds: Set, - fenceChanges: [string, boolean][] - ): void { - const blockedPaneKeys = new Set(plan.blockedPanes.map((blocked) => blocked.paneKey)) - for (const paneKey of this.announcedBlockedPaneKeys) { - if (!blockedPaneKeys.has(paneKey)) { - this.announcedBlockedPaneKeys.delete(paneKey) - fenceChanges.push([paneKey, false]) - } - } - for (const hostId of store.getWorkspaceSessionHostIds?.() ?? [LOCAL_EXECUTION_HOST_ID]) { - const staged = sessions.get(hostId) - const session = staged?.next ?? store.getWorkspaceSession?.(hostId) - const retired = Object.entries(session?.sleepingAgentSessionsByPaneKey ?? {}).filter( - ([paneKey, record]) => - record.automaticResumeBlockedBy === 'legacy-orchestration-worker' && - !blockedPaneKeys.has(paneKey) - ) - if (retired.length === 0) { - continue - } - let state = staged - if (!state) { - const current = store.getWorkspaceSession?.(hostId) - if (!current) { - continue - } - state = { current, next: structuredClone(current) } - sessions.set(hostId, state) - } - const next = { ...state.next.sleepingAgentSessionsByPaneKey } - for (const [paneKey, record] of retired) { - const { automaticResumeBlockedBy: _retired, ...unfenced } = record - next[paneKey] = unfenced - fenceChanges.push([paneKey, false]) - } - state.next.sleepingAgentSessionsByPaneKey = next - changedHostIds.add(hostId) - } + return this.getPlan() ?? { candidates: [], ambiguousDispatchIds: [] } } async persist( diff --git a/src/main/runtime/runtime-legacy-worker-terminal-recovery-runner.ts b/src/main/runtime/runtime-legacy-worker-terminal-recovery-runner.ts index bd15abc7d4d..6bbe3b2ed2f 100644 --- a/src/main/runtime/runtime-legacy-worker-terminal-recovery-runner.ts +++ b/src/main/runtime/runtime-legacy-worker-terminal-recovery-runner.ts @@ -104,7 +104,6 @@ export async function runLegacyWorkerTerminalRecovery( exitedDispatchIds.push(candidate.dispatchId) } const result = { - blockedPaneCount: plan.blockedPanes.length, adoptedDispatchIds, exitedDispatchIds, deferredDispatchIds: [...deferredDispatchIds] diff --git a/src/main/runtime/runtime-legacy-worker-terminal-recovery-types.ts b/src/main/runtime/runtime-legacy-worker-terminal-recovery-types.ts index c65afd73952..16ca330647c 100644 --- a/src/main/runtime/runtime-legacy-worker-terminal-recovery-types.ts +++ b/src/main/runtime/runtime-legacy-worker-terminal-recovery-types.ts @@ -5,7 +5,6 @@ import type { PtyControllerInventory } from './runtime-pty-controller-contract' import type { ResolvedWorktree } from './runtime-worktree-path-identity' export type LegacyWorkerTerminalRecoveryResult = { - blockedPaneCount: number adoptedDispatchIds: string[] exitedDispatchIds: string[] deferredDispatchIds: string[] diff --git a/src/main/runtime/runtime-legacy-worker-terminal-resume-fence.test.ts b/src/main/runtime/runtime-legacy-worker-terminal-resume-fence.test.ts deleted file mode 100644 index 6fe14f2abe7..00000000000 --- a/src/main/runtime/runtime-legacy-worker-terminal-resume-fence.test.ts +++ /dev/null @@ -1,286 +0,0 @@ -import { afterEach, describe, expect, it, vi } from 'vitest' -import { getDefaultWorkspaceSession } from '../../shared/constants' -import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' -import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' -import { OrchestrationDb } from './orchestration/db' -import { OrcaRuntimeService } from './orca-runtime' -import { ORCHESTRATION_METHODS } from './rpc/methods/orchestration' -import { RuntimeLegacyWorkerTerminalRecoveryPersistence } from './runtime-legacy-worker-terminal-recovery-persistence' -import type { RuntimeStore } from './runtime-store-contract' - -const PANE_KEY = 'tab_worker:33333333-3333-4333-8333-333333333333' -const WORKTREE_ID = 'repo::worktree' - -function sessionWithSleepingWorker(): WorkspaceSessionState { - return { - ...getDefaultWorkspaceSession(), - sleepingAgentSessionsByPaneKey: { - [PANE_KEY]: { - paneKey: PANE_KEY, - tabId: 'tab_worker', - worktreeId: WORKTREE_ID, - agent: 'codex', - providerSession: { key: 'session_id', id: 'codex-session-1' }, - prompt: '', - state: 'done', - capturedAt: 1, - updatedAt: 1, - origin: 'live' - } - } - } as WorkspaceSessionState -} - -describe('settled worker automatic-resume fence persistence', () => { - let db: OrchestrationDb | undefined - - afterEach(() => db?.close()) - - function harness( - onFenceChanged?: (paneKey: string, blocked: boolean) => void, - /** False models a worker that settles while its tab is still open: no record to stamp yet. */ - withSleepingRecord = true - ): { - db: OrchestrationDb - taskId: string - dispatchId: string - persistence: RuntimeLegacyWorkerTerminalRecoveryPersistence - fence: () => string | undefined - } { - const orchestrationDb = new OrchestrationDb(':memory:') - db = orchestrationDb - let session = withSleepingRecord - ? sessionWithSleepingWorker() - : (getDefaultWorkspaceSession() as WorkspaceSessionState) - const store = { - getWorkspaceSession: () => session, - setWorkspaceSession: (next: WorkspaceSessionState) => { - session = next - }, - getWorkspaceSessionHostIds: () => [LOCAL_EXECUTION_HOST_ID], - flushOrThrow: vi.fn() - } as unknown as RuntimeStore - const task = orchestrationDb.createTask({ spec: 'fence me' }) - const started = orchestrationDb.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: task.id, - startOptions: {} - }) - orchestrationDb.prepareStartingWorkerAuthority({ - dispatchId: started.dispatch.id, - handle: 'term_worker', - paneKey: PANE_KEY, - processIncarnation: 'runtime:pty:1', - worktreeId: WORKTREE_ID, - setupState: 'not_applicable', - effects: [], - terminalOwnership: 'created' - }) - orchestrationDb.markWorkerDispatchReady(started.dispatch.id) - return { - db: orchestrationDb, - taskId: task.id, - dispatchId: started.dispatch.id, - persistence: new RuntimeLegacyWorkerTerminalRecoveryPersistence( - () => store, - () => orchestrationDb, - () => LOCAL_EXECUTION_HOST_ID, - onFenceChanged - ), - fence: () => session.sleepingAgentSessionsByPaneKey?.[PANE_KEY]?.automaticResumeBlockedBy - } - } - - function settle(d: OrchestrationDb, taskId: string, dispatchId: string): void { - expect( - d.settleWorkerReport({ taskId, dispatchId, outcome: 'succeeded', result: 'done' }).action - ).toBe('settled') - } - - it('pushes the fence to the live renderer instead of waiting for the next app start', () => { - const fenceChanges: [string, boolean][] = [] - const h = harness((paneKey, blocked) => fenceChanges.push([paneKey, blocked])) - settle(h.db, h.taskId, h.dispatchId) - - h.persistence.prepare() - - expect(fenceChanges).toEqual([[PANE_KEY, true]]) - }) - - it('announces the fence for a pane that has no sleeping record to stamp yet', () => { - const fenceChanges: [string, boolean][] = [] - const h = harness((paneKey, blocked) => fenceChanges.push([paneKey, blocked]), false) - settle(h.db, h.taskId, h.dispatchId) - - h.persistence.prepare() - expect(fenceChanges).toEqual([[PANE_KEY, true]]) - - const requested = h.db.requestWorkerTerminalRelease(h.dispatchId) - h.db.settleWorkerTerminalRelease((requested as { resource: { id: string } }).resource.id) - h.persistence.prepare() - - // A fence the plan no longer claims must be lifted even with no record to read it from. - expect(fenceChanges).toEqual([ - [PANE_KEY, true], - [PANE_KEY, false] - ]) - }) - - // The STA-4577 repro: worker_done, no release, restart, open the worktree — the pane still - // holds a resumable provider session and must not respawn `codex resume`. - it('fences a settled worker pane whose terminal was never released', () => { - const h = harness() - settle(h.db, h.taskId, h.dispatchId) - - h.persistence.prepare() - - expect(h.fence()).toBe('legacy-orchestration-worker') - }) - - it('lifts the fence once release retires the terminal resource', () => { - const h = harness() - settle(h.db, h.taskId, h.dispatchId) - h.persistence.prepare() - expect(h.fence()).toBe('legacy-orchestration-worker') - - const requested = h.db.requestWorkerTerminalRelease(h.dispatchId) - expect(requested.disposition).toBe('requested') - h.db.settleWorkerTerminalRelease((requested as { resource: { id: string } }).resource.id) - h.persistence.prepare() - - expect(h.fence()).toBeUndefined() - }) - - it('lifts the fence when the user takes the pane over', () => { - const h = harness() - settle(h.db, h.taskId, h.dispatchId) - h.persistence.prepare() - expect(h.fence()).toBe('legacy-orchestration-worker') - - expect(h.db.markWorkerTerminalUserOwned(PANE_KEY)).toBe(1) - h.persistence.prepare() - - expect(h.fence()).toBeUndefined() - }) - - // An unreadable plan is not evidence a pane stopped needing its fence. - it('keeps the fence when the recovery plan cannot be read', () => { - const h = harness() - settle(h.db, h.taskId, h.dispatchId) - h.persistence.prepare() - expect(h.fence()).toBe('legacy-orchestration-worker') - - vi.spyOn(h.db, 'listLegacyWorkerTerminalRecoveryRows').mockImplementation(() => { - throw new Error('orchestration_db_unavailable') - }) - const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) - try { - expect(h.persistence.prepare()).toEqual({ - blockedPanes: [], - candidates: [], - ambiguousDispatchIds: [] - }) - } finally { - warn.mockRestore() - } - - expect(h.fence()).toBe('legacy-orchestration-worker') - }) - - // A live worker's pane was already fenced while main reconciles it against PTY inventory; the - // settled arm must not disturb that, and the plan must still name it as unsettled. - it('keeps a live worker pane fenced and marked unsettled', () => { - const h = harness() - - const plan = h.persistence.prepare() - - expect(h.fence()).toBe('legacy-orchestration-worker') - expect(plan.blockedPanes).toEqual([ - expect.objectContaining({ paneKey: PANE_KEY, settled: false }) - ]) - expect(plan.candidates).toEqual([expect.objectContaining({ dispatchId: h.dispatchId })]) - }) -}) - -// STA-4577's other half: settlement with no release and no restart. The stamp only ran at startup -// and after release/retain/takeover, so reopening the pane in the same session respawned the agent. -describe('worker_done without a release', () => { - let db: OrchestrationDb | undefined - - afterEach(() => db?.close()) - - it('fences the pane in the same session', async () => { - const orchestrationDb = new OrchestrationDb(':memory:') - db = orchestrationDb - let session = sessionWithSleepingWorker() - const store = { - getWorkspaceSession: () => session, - setWorkspaceSession: (next: WorkspaceSessionState) => { - session = next - }, - getWorkspaceSessionHostIds: () => [LOCAL_EXECUTION_HOST_ID], - flushOrThrow: vi.fn() - } as unknown as RuntimeStore - const runtime = new OrcaRuntimeService(store) - runtime.setOrchestrationDb(orchestrationDb) - vi.spyOn(runtime, 'getTerminalPaneKey').mockImplementation((handle) => - handle === 'term_worker' ? PANE_KEY : 'tab_coord:aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa' - ) - vi.spyOn(runtime, 'getTerminalProcessIncarnation').mockReturnValue('runtime:pty:1') - vi.spyOn(runtime, 'notifyMessageArrived').mockImplementation(() => {}) - - const run = orchestrationDb.createRun({ - objective: 'settle without release', - coordinatorHandle: 'term_coord', - coordinatorPaneKey: 'tab_coord:aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa' - }) - const task = orchestrationDb.createTask({ spec: 'settle without release', runId: run.id }) - const started = orchestrationDb.createStartingWorkerDispatch({ - creator: { kind: 'system' }, - maxDepth: Number.MAX_SAFE_INTEGER, - taskId: task.id, - startOptions: {} - }) - orchestrationDb.prepareStartingWorkerAuthority({ - dispatchId: started.dispatch.id, - handle: 'term_worker', - paneKey: PANE_KEY, - processIncarnation: 'runtime:pty:1', - worktreeId: WORKTREE_ID, - setupState: 'not_applicable', - effects: [], - terminalOwnership: 'created' - }) - orchestrationDb.markWorkerDispatchReady(started.dispatch.id) - const capability = orchestrationDb.mintDispatchCapability({ - dispatchId: started.dispatch.id, - paneKey: PANE_KEY, - processIncarnation: 'runtime:pty:1' - }) - expect(session.sleepingAgentSessionsByPaneKey?.[PANE_KEY]?.automaticResumeBlockedBy).toBe( - undefined - ) - - const send = ORCHESTRATION_METHODS.find((method) => method.name === 'orchestration.send')! - await send.handler( - send.params!.parse({ - from: 'term_worker', - to: 'term_coord', - subject: 'Done', - type: 'worker_done', - payload: JSON.stringify({ - taskId: task.id, - dispatchId: started.dispatch.id, - outcome: 'succeeded' - }) - }), - { runtime, orchestrationCapability: capability } - ) - - expect(orchestrationDb.getWorkerDispatch(started.dispatch.id)?.state).toBe('succeeded') - expect(session.sleepingAgentSessionsByPaneKey?.[PANE_KEY]?.automaticResumeBlockedBy).toBe( - 'legacy-orchestration-worker' - ) - }) -}) diff --git a/src/main/runtime/runtime-mobile-session-result-finalization.ts b/src/main/runtime/runtime-mobile-session-result-finalization.ts index 2c3af07b6f9..1848ae56793 100644 --- a/src/main/runtime/runtime-mobile-session-result-finalization.ts +++ b/src/main/runtime/runtime-mobile-session-result-finalization.ts @@ -1,4 +1,5 @@ import type { RuntimeMobileSessionTabsResult } from '../../shared/runtime-types' +import { dropRetirementProofsForLiveSurfaces } from './mobile-session-terminal-retirement-proof' import type { RuntimeMobileSessionProjectionHost, RuntimeMobileSessionProjectionInput @@ -41,14 +42,9 @@ export function finalizeRuntimeMobileSessionTabsResult( ...(snapshot.tabGroupLayout !== undefined ? { tabGroupLayout } : {}), ...(snapshot.retiredTerminalSurfaces ? { - retiredTerminalSurfaces: snapshot.retiredTerminalSurfaces.filter( - (retired) => - !snapshot.tabs.some( - (tab) => - tab.type === 'terminal' && - tab.parentTabId === retired.parentTabId && - tab.leafId === retired.leafId - ) + retiredTerminalSurfaces: dropRetirementProofsForLiveSurfaces( + snapshot.retiredTerminalSurfaces, + snapshot.tabs ) } : {}), diff --git a/src/main/runtime/runtime-notifier-contract.ts b/src/main/runtime/runtime-notifier-contract.ts index aa2982082b4..9cdc365e1ba 100644 --- a/src/main/runtime/runtime-notifier-contract.ts +++ b/src/main/runtime/runtime-notifier-contract.ts @@ -80,7 +80,6 @@ export type RuntimeNotifier = { ptyId?: string ): void /** The fence lives in the workspace session, which a live renderer only re-reads at startup. */ - setLegacyWorkerTerminalResumeFence?(paneKey: string, blocked: boolean): void splitTerminal( tabId: string, paneRuntimeId: number, diff --git a/src/main/runtime/runtime-rpc-request-authorization.test.ts b/src/main/runtime/runtime-rpc-request-authorization.test.ts index cf1af0af90a..d5a083a56bf 100644 --- a/src/main/runtime/runtime-rpc-request-authorization.test.ts +++ b/src/main/runtime/runtime-rpc-request-authorization.test.ts @@ -93,9 +93,19 @@ describe('OrcaRuntimeRpcServer', () => { } try { - db.insertMessage({ from: 'worker', to: 'coordinator', subject: 'before reset' }) + db.insertMessage({ + runId: 'run_legacy_local', + from: 'worker', + to: 'coordinator', + subject: 'before reset' + }) const first = await resetMessages('reset-first', firstDevice.token) - db.insertMessage({ from: 'worker', to: 'coordinator', subject: 'after reset' }) + db.insertMessage({ + runId: 'run_legacy_local', + from: 'worker', + to: 'coordinator', + subject: 'after reset' + }) const replay = await resetMessages('reset-replay', firstDevice.token) expect(first).toMatchObject({ @@ -130,6 +140,7 @@ describe('OrcaRuntimeRpcServer', () => { const device = server['deviceRegistry']!.addDevice('existing-cli', 'runtime') const existingFingerprint = createHash('sha256').update(device.token).digest('hex') db.createRemoteDispatchAttachment({ + runId: 'run_home', dispatchId: 'ctx_existing_remote', taskId: 'task_existing_remote', homePeerFingerprint: existingFingerprint, diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts b/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts index 92033ef0827..ca8d326e7e7 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts @@ -248,6 +248,8 @@ export const MOBILE_RPC_METHOD_ALLOWLIST = new Set([ 'status.get', 'agentTeams.prepareLaunch', 'agentTeams.tmuxCompat', + // Why: the phone reports a takeover out of band, the same signal the desktop renderer sends. + 'orchestration.workerTerminalUserInput', 'terminal.clearBuffer', 'terminal.close', 'terminal.closeAll', diff --git a/src/main/runtime/runtime-terminal-orphan-topology-validation.test.ts b/src/main/runtime/runtime-terminal-orphan-topology-validation.test.ts new file mode 100644 index 00000000000..cb8f22e864b --- /dev/null +++ b/src/main/runtime/runtime-terminal-orphan-topology-validation.test.ts @@ -0,0 +1,151 @@ +import { expect, it } from 'vitest' +import type { RuntimeTerminalOrphanAdoptionRequest } from '../../shared/runtime-types' +import { validateRuntimeTerminalOrphanTopology } from './runtime-terminal-orphan-topology-validation' + +function fixture(count: number): RuntimeTerminalOrphanAdoptionRequest { + const ids = Array.from({ length: count }, (_, index) => `tab-${index}`) + return { + worktree: 'folder-workspace', + expectedTopologyRevision: 1, + claims: ids.map((tabId) => ({ + tabId, + leafId: tabId, + terminal: tabId, + ptyId: tabId, + incarnationId: tabId + })) as RuntimeTerminalOrphanAdoptionRequest['claims'], + topology: { + tabs: ids.map((tabId) => ({ + tabId, + root: { type: 'leaf', leafId: tabId }, + activeLeafId: tabId, + expandedLeafId: null + })), + groups: [{ id: 'g', activeTabId: ids[0], tabOrder: ids, recentTabIds: ids.toReversed() }] + } + } +} + +it('validates large restored MRU lists with linear tab-order reads', () => { + const request = fixture(1000) + let reads = 0 + const group = request.topology!.groups[0] + group.tabOrder = new Proxy(group.tabOrder, { + get(target, key, receiver) { + if (typeof key === 'string' && /^\d+$/.test(key)) { + reads += 1 + } + return Reflect.get(target, key, receiver) + } + }) + expect( + validateRuntimeTerminalOrphanTopology( + request, + request.claims.map((claim) => ({ claim })) + ).topologyTabsById.size + ).toBe(1000) + expect(reads).toBeLessThanOrEqual(3000) +}) + +it.each(['duplicate', 'foreign-recent', 'foreign-active'])( + 'rejects %s group membership', + (kind) => { + const request = fixture(2) + const group = request.topology!.groups[0] + if (kind === 'duplicate') { + group.tabOrder.push(group.tabOrder[0]) + } + if (kind === 'foreign-recent') { + group.recentTabIds = ['foreign'] + } + if (kind === 'foreign-active') { + group.activeTabId = 'foreign' + } + expect(() => + validateRuntimeTerminalOrphanTopology( + request, + request.claims.map((claim) => ({ claim })) + ) + ).toThrow('terminal_orphan_topology_invalid') + } +) + +function validate(request: RuntimeTerminalOrphanAdoptionRequest) { + return validateRuntimeTerminalOrphanTopology( + request, + request.claims.map((claim) => ({ claim })) + ) +} + +type Groups = NonNullable['groups'] + +function withGroups(count: number, groups: Groups): RuntimeTerminalOrphanAdoptionRequest { + const request = fixture(count) + request.topology!.groups = groups + return request +} + +// Membership is per-group, but the no-tab-in-two-groups rule is global. Replacing that rule with +// the per-group set would let one pane be adopted into two groups and cross-wire the session. +it('rejects a tab claimed by two different groups', () => { + expect(() => + validate( + withGroups(2, [ + { id: 'g1', activeTabId: 'tab-0', tabOrder: ['tab-0', 'tab-1'], recentTabIds: [] }, + { id: 'g2', activeTabId: 'tab-1', tabOrder: ['tab-1'], recentTabIds: [] } + ]) + ) + ).toThrow('terminal_orphan_topology_invalid') +}) + +it('rejects a duplicated group id', () => { + expect(() => + validate( + withGroups(2, [ + { id: 'g', activeTabId: 'tab-0', tabOrder: ['tab-0'], recentTabIds: [] }, + { id: 'g', activeTabId: 'tab-1', tabOrder: ['tab-1'], recentTabIds: [] } + ]) + ) + ).toThrow('terminal_orphan_topology_invalid') +}) + +// Cardinality 0: an empty tab order can never hold the active tab, so adoption must fail closed +// rather than fall through to an arbitrary pane. +it('rejects an empty tab order', () => { + expect(() => + validate(withGroups(2, [{ id: 'g', activeTabId: 'tab-0', tabOrder: [], recentTabIds: [] }])) + ).toThrow('terminal_orphan_topology_invalid') +}) + +it('rejects a claimed tab that no group lists', () => { + expect(() => + validate( + withGroups(2, [{ id: 'g', activeTabId: 'tab-0', tabOrder: ['tab-0'], recentTabIds: [] }]) + ) + ).toThrow('terminal_orphan_topology_invalid') +}) + +it.each([ + ['1 tab per group', [['tab-0'], ['tab-1']]], + ['both tabs in one group', [['tab-0', 'tab-1']]] +])('accepts an exactly-covering split with %s', (_label, tabOrders) => { + const groups: Groups = tabOrders.map((tabOrder, i) => ({ + id: `g${i}`, + activeTabId: tabOrder[0], + tabOrder, + // Reverse MRU: every entry must still resolve inside its own group. + recentTabIds: tabOrder.toReversed() + })) + expect(validate(withGroups(2, groups)).topologyTabsById.size).toBe(2) +}) + +it.each([ + ['omitted', undefined], + ['empty', [] as string[]] +])('accepts %s recentTabIds', (_label, recentTabIds) => { + expect( + validate( + withGroups(2, [{ id: 'g', activeTabId: 'tab-0', tabOrder: ['tab-0', 'tab-1'], recentTabIds }]) + ).topologyTabsById.size + ).toBe(2) +}) diff --git a/src/main/runtime/runtime-terminal-orphan-topology-validation.ts b/src/main/runtime/runtime-terminal-orphan-topology-validation.ts index 72b1e43ef08..0864533f4c7 100644 --- a/src/main/runtime/runtime-terminal-orphan-topology-validation.ts +++ b/src/main/runtime/runtime-terminal-orphan-topology-validation.ts @@ -54,7 +54,8 @@ export function validateRuntimeTerminalOrphanTopology( const seenGroupIds = new Set() const groupedTabIds = new Set() for (const group of topologyGroups) { - if (seenGroupIds.has(group.id) || !group.tabOrder.includes(group.activeTabId)) { + const groupTabIds = new Set(group.tabOrder) + if (seenGroupIds.has(group.id) || !groupTabIds.has(group.activeTabId)) { throw new Error('terminal_orphan_topology_invalid') } seenGroupIds.add(group.id) @@ -64,7 +65,7 @@ export function validateRuntimeTerminalOrphanTopology( } groupedTabIds.add(tabId) } - if (group.recentTabIds?.some((tabId) => !group.tabOrder.includes(tabId))) { + if (group.recentTabIds?.some((tabId) => !groupTabIds.has(tabId))) { throw new Error('terminal_orphan_topology_invalid') } } diff --git a/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts b/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts new file mode 100644 index 00000000000..84966846be7 --- /dev/null +++ b/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts @@ -0,0 +1,120 @@ +import { collectRuntimeWorktreeAgentSources } from './runtime-worktree-agent-sources' +import { describe, expect, it } from 'vitest' +import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' +import { + structuredAgentSessionPaneKey, + structuredAgentSessionTabId +} from '../../shared/structured-agent-session-projection' +import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' +import type { RuntimeWorktreePsSummary } from '../../shared/runtime-types' + +/** + * A structured session has no PTY, so it reaches none of the hook or retained snapshots that every + * other row comes from. Before this, `worktree ps` reported a worktree running one as idle while + * the sidebar showed it working — the CLI, which is the agent-facing surface, was the blind one. + */ +const WORKTREE_ID = 'repo-1::/workspace/app' + +function summary(over: Partial = {}): AgentSessionStatusSummary { + return { + sessionId: 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d', + workspaceId: WORKTREE_ID, + agent: 'claude', + status: 'working', + latestPrompt: 'ship the thing', + updatedAt: 1_757_030_400_000, + hostExecutionOwned: true, + ...over + } as AgentSessionStatusSummary +} + +function attach(summaries: AgentSessionStatusSummary[]): RuntimeWorktreePsSummary { + const row = { + worktreeId: WORKTREE_ID, + status: 'inactive', + hasHostSidebarActivity: false, + agents: [] + } as unknown as RuntimeWorktreePsSummary + const summariesById = new Map([[WORKTREE_ID, row]]) + attachRuntimeWorktreeAgentRows({ + summaries: summariesById, + pathIndex: { byPath: new Map(), byRealPath: new Map() } as never, + missingWorktreeIds: new Set(), + workingTerminalEvidenceByWorktreeId: new Map(), + rowSources: collectRuntimeWorktreeAgentSources({ + mirroredWorktreeIdByTabId: new Map(), + connectedPtyEvidence: { tabIds: new Set(), paneKeys: new Set(), ptyIds: new Set() }, + retainedSnapshots: [], + hookSnapshots: [], + structuredSummaries: summaries + }), + orchestrationByPaneKey: null, + getSummary: (map, _p, _m, id) => map.get(id) ?? null + }) + return row +} + +describe('worktree ps reports structured sessions', () => { + it('a busy structured session is not reported idle', () => { + const row = attach([summary()]) + expect(row.agents).toHaveLength(1) + expect(row.agents[0]?.state).toBe('working') + expect(row.agents[0]?.agentType).toBe('claude') + expect(row.agents[0]?.prompt).toBe('ship the thing') + }) + + // The same projection the sidebar applies, so the two surfaces cannot disagree about one session. + it('maps attention to blocked and idle to done', () => { + expect(attach([summary({ status: 'attention' })]).agents[0]?.state).toBe('blocked') + expect(attach([summary({ status: 'idle' })]).agents[0]?.state).toBe('done') + }) + + it('does not turn a completed host-held session into permission', () => { + const row = attach([summary({ status: 'idle' })]) + expect(row.status).toBe('inactive') + expect(row.hasHostSidebarActivity).toBe(false) + }) + + it('reports the DERIVED pane key, never an orchestration credential', () => { + const row = attach([summary()]) + const sessionId = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' + expect(row.agents[0]?.paneKey).toBe( + structuredAgentSessionPaneKey(structuredAgentSessionTabId(sessionId), sessionId) + ) + }) + + // Null status means no turn has been persisted; the chat itself shows nothing, so neither does this. + it('omits a session with no projected status', () => { + expect(attach([summary({ status: null })]).agents).toHaveLength(0) + }) +}) + +/** + * The deliberate non-goal. Adding structured rows to `terminal list` was investigated and rejected: + * mobile mounts a terminal WebView per row that can never receive a frame, a `connected`-keyed + * refresh check goes permanently true and pins shipped clients to a fast cadence with no exit, and + * the plugin projection has no field that can carry `writable: false`. Every SAFE consumer of a + * terminal summary checks `ptyId`; the breaking ones key off `connected` or mere row presence, + * which no added field can qualify. A separate change publishes an honest partial-listing count + * there instead. This pins that only `worktree ps` gained the enumerator. + */ +describe('terminal listing is deliberately left alone', () => { + it('only worktree ps consumes the structured status summaries', async () => { + const { readFile } = await import('node:fs/promises') + // orca-runtime-subscribe-to-terminal-resize.ts owns listTerminals. + const listing = await readFile( + new URL('./orca-runtime-subscribe-to-terminal-resize.ts', import.meta.url), + 'utf8' + ) + // Guard the guard: an empty read would make every assertion below vacuously true. + expect(listing).toContain('async listTerminals(') + expect(listing).not.toContain('liveSessionStatusSummaries') + expect(listing).not.toContain('structuredSummaries') + + const worktreePs = await readFile( + new URL('./orca-runtime-get-worktree-ps.ts', import.meta.url), + 'utf8' + ) + expect(worktreePs).toContain('liveSessionStatusSummaries') + }) +}) diff --git a/src/main/runtime/runtime-worktree-agent-rows.ts b/src/main/runtime/runtime-worktree-agent-rows.ts index 1853f704471..80802cf11a3 100644 --- a/src/main/runtime/runtime-worktree-agent-rows.ts +++ b/src/main/runtime/runtime-worktree-agent-rows.ts @@ -1,34 +1,10 @@ -import { - AGENT_STATUS_STALE_AFTER_MS, - isFreshNonDoneAgentStatus, - pickParsedAgentStatusPayload, - type AgentStatusIpcPayload, - type ParsedAgentStatusPayload -} from '../../shared/agent-status-types' -import { terminalStatusPayloadMatchesHook } from '../../shared/agent-terminal-status-equivalence' +import { isFreshNonDoneAgentStatus } from '../../shared/agent-status-types' import type { RuntimeWorktreeAgentRow, RuntimeWorktreePsSummary } from '../../shared/runtime-types' -import { parseLegacyNumericPaneKey, parsePaneKey } from '../../shared/stable-pane-id' -import { isWslHookRelayConnectionId } from '../../shared/wsl-hook-relay-contract' import { mergeWorktreeSummaryStatus } from './runtime-worktree-status-projection' import type { RuntimeWorktreeSummaryPathIndex } from './runtime-worktree-summary-paths' import type { RuntimeWorkingTerminalEvidence } from './runtime-worktree-ps-activity' - -export type RuntimeAgentRowSnapshot = { - paneKey: string - ptyId: string - worktreeId?: string - tabId?: string - connectionId: string | null - payload: ParsedAgentStatusPayload - stateStartedAt: number - updatedAt: number -} - -type ConnectedPtyEvidence = { - tabIds: ReadonlySet - paneKeys: ReadonlySet - ptyIds: ReadonlySet -} +import type { RuntimeWorktreeAgentSource } from './runtime-worktree-agent-source' +export type { RuntimeAgentRowSnapshot } from './runtime-worktree-pty-agent-sources' type OrchestrationDisplay = { taskTitle?: string | null @@ -36,38 +12,15 @@ type OrchestrationDisplay = { parentPaneKey?: string | null } -type RuntimeWorktreeAgentSource = { - paneKey: string - ptyId?: string - tabId?: string - worktreeId?: string - connectionId: string | null - payload: ParsedAgentStatusPayload - state: ParsedAgentStatusPayload['state'] - workingMode?: ParsedAgentStatusPayload['workingMode'] - agentType: string | null - prompt: string - lastAssistantMessage: string | null - toolName: string | null - toolInput: string | null - interrupted: boolean - stateStartedAt: number - updatedAt: number - restoredUnconfirmed?: boolean -} - export function attachRuntimeWorktreeAgentRows(args: { summaries: Map pathIndex: RuntimeWorktreeSummaryPathIndex missingWorktreeIds: Set - mirroredWorktreeIdByTabId: ReadonlyMap - connectedPtyEvidence: ConnectedPtyEvidence + rowSources: ReadonlyMap workingTerminalEvidenceByWorktreeId: ReadonlyMap< string, readonly RuntimeWorkingTerminalEvidence[] > - retainedSnapshots: Iterable - hookSnapshots: readonly AgentStatusIpcPayload[] orchestrationByPaneKey: Record | null | undefined getSummary: ( summaries: Map, @@ -76,88 +29,11 @@ export function attachRuntimeWorktreeAgentRows(args: { worktreeId: string ) => RuntimeWorktreePsSummary | null }): void { - const rowSources = new Map() + const { rowSources } = args const now = Date.now() - for (const snapshot of args.retainedSnapshots) { - const { payload } = snapshot - rowSources.set(snapshot.paneKey, { - paneKey: snapshot.paneKey, - ptyId: snapshot.ptyId, - tabId: snapshot.tabId, - worktreeId: snapshot.worktreeId, - connectionId: snapshot.connectionId, - payload, - state: payload.state, - ...(payload.workingMode ? { workingMode: payload.workingMode } : {}), - agentType: payload.agentType ?? null, - prompt: payload.prompt, - lastAssistantMessage: payload.lastAssistantMessage ?? null, - toolName: payload.toolName ?? null, - toolInput: payload.toolInput ?? null, - interrupted: payload.interrupted ?? false, - stateStartedAt: snapshot.stateStartedAt, - updatedAt: snapshot.updatedAt - }) - } - for (const entry of args.hookSnapshots) { - if (entry.restoredUnconfirmed === true) { - continue - } - const existing = rowSources.get(entry.paneKey) - const hookPayload = pickParsedAgentStatusPayload(entry) - if (existing && existing.updatedAt > entry.receivedAt) { - if ( - entry.workingMode === 'monitoring' && - now - entry.receivedAt <= AGENT_STATUS_STALE_AFTER_MS && - terminalStatusPayloadMatchesHook(hookPayload, existing.payload) - ) { - existing.workingMode = 'monitoring' - if (existing.payload.workingMode === undefined) { - existing.payload = { ...existing.payload, workingMode: 'monitoring' } - } - } - continue - } - rowSources.set(entry.paneKey, { - paneKey: entry.paneKey, - ptyId: existing?.ptyId, - tabId: entry.tabId, - worktreeId: entry.worktreeId, - connectionId: entry.connectionId, - payload: hookPayload, - state: entry.state, - ...(entry.workingMode ? { workingMode: entry.workingMode } : {}), - agentType: entry.agentType ?? null, - prompt: entry.prompt, - lastAssistantMessage: entry.lastAssistantMessage ?? null, - toolName: entry.toolName ?? null, - toolInput: entry.toolInput ?? null, - interrupted: entry.interrupted ?? false, - stateStartedAt: entry.stateStartedAt, - updatedAt: entry.receivedAt - }) - } - if (rowSources.size === 0) { - return - } const rowsByWorktree = new Map() for (const source of rowSources.values()) { - const tabId = - source.tabId ?? - parsePaneKey(source.paneKey)?.tabId ?? - parseLegacyNumericPaneKey(source.paneKey)?.tabId - const mirroredWorktreeId = tabId ? args.mirroredWorktreeIdByTabId.get(tabId) : undefined - if ( - tabId !== undefined && - mirroredWorktreeId === undefined && - (source.connectionId === null || isWslHookRelayConnectionId(source.connectionId)) && - !args.connectedPtyEvidence.tabIds.has(tabId) && - !args.connectedPtyEvidence.paneKeys.has(source.paneKey) && - (source.ptyId === undefined || !args.connectedPtyEvidence.ptyIds.has(source.ptyId)) - ) { - continue - } - const worktreeId = mirroredWorktreeId ?? source.worktreeId + const { worktreeId } = source if (!worktreeId) { continue } @@ -204,13 +80,15 @@ export function attachRuntimeWorktreeAgentRows(args: { let hasForegroundWorkingAgent = false const monitoringSources: RuntimeWorktreeAgentSource[] = [] for (const row of rows) { - if (!isFreshNonDoneAgentStatus(row, now)) { + const source = rowSources.get(row.paneKey) + const hostHeldStructuredSession = + source?.authority === 'structured-host' && row.state !== 'done' + if (!hostHeldStructuredSession && !isFreshNonDoneAgentStatus(row, now)) { continue } summary.hasHostSidebarActivity = true if (row.state === 'working') { if (row.workingMode === 'monitoring') { - const source = rowSources.get(row.paneKey) if (source) { monitoringSources.push(source) } diff --git a/src/main/runtime/runtime-worktree-agent-source.ts b/src/main/runtime/runtime-worktree-agent-source.ts new file mode 100644 index 00000000000..2e8bc833b23 --- /dev/null +++ b/src/main/runtime/runtime-worktree-agent-source.ts @@ -0,0 +1,21 @@ +import type { ParsedAgentStatusPayload } from '../../shared/agent-status-types' + +export type RuntimeWorktreeAgentSource = { + paneKey: string + ptyId?: string + tabId?: string + worktreeId?: string + connectionId: string | null + state: ParsedAgentStatusPayload['state'] + workingMode?: ParsedAgentStatusPayload['workingMode'] + agentType: string | null + prompt: string + lastAssistantMessage: string | null + toolName: string | null + toolInput: string | null + interrupted: boolean + stateStartedAt: number + updatedAt: number + /** Structured host projections remain authoritative after PTY freshness expiry. */ + authority?: 'structured-host' +} diff --git a/src/main/runtime/runtime-worktree-agent-sources.test.ts b/src/main/runtime/runtime-worktree-agent-sources.test.ts new file mode 100644 index 00000000000..c0395cd6831 --- /dev/null +++ b/src/main/runtime/runtime-worktree-agent-sources.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it } from 'vitest' +import { collectRuntimeWorktreeAgentSources } from './runtime-worktree-agent-sources' +import type { RuntimeAgentRowSnapshot } from './runtime-worktree-pty-agent-sources' +import type { AgentStatusIpcPayload } from '../../shared/agent-status-types' + +const paneKey = 'worktree:tab:0' +const now = Date.now() +const retained: RuntimeAgentRowSnapshot = { + paneKey, + ptyId: 'pty', + tabId: 'tab', + worktreeId: 'worktree', + connectionId: null, + payload: { state: 'working', prompt: 'implement', agentType: 'codex' }, + stateStartedAt: now, + updatedAt: now +} +const base = { + retainedSnapshots: [retained], + hookSnapshots: [] as AgentStatusIpcPayload[], + structuredSummaries: [], + mirroredWorktreeIdByTabId: new Map(), + connectedPtyEvidence: { + tabIds: new Set(), + paneKeys: new Set(), + ptyIds: new Set() + } +} + +describe('worktree agent source admission', () => { + it('rejects a disconnected local terminal before row assembly', () => { + expect(collectRuntimeWorktreeAgentSources(base).size).toBe(0) + const connected = { + ...base, + connectedPtyEvidence: { ...base.connectedPtyEvidence, ptyIds: new Set(['pty']) } + } + expect(collectRuntimeWorktreeAgentSources(connected).get(paneKey)?.state).toBe('working') + }) + + it('keeps remote evidence and resolves mirrored workspace ownership', () => { + const remote = { ...retained, connectionId: 'ssh-connection' } + expect(collectRuntimeWorktreeAgentSources({ ...base, retainedSnapshots: [remote] }).size).toBe( + 1 + ) + const sources = collectRuntimeWorktreeAgentSources({ + ...base, + mirroredWorktreeIdByTabId: new Map([['tab', 'remote-worktree']]) + }) + expect(sources.get(paneKey)?.worktreeId).toBe('remote-worktree') + }) + + it('preserves fresh monitoring enrichment on a newer retained report', () => { + const hook: AgentStatusIpcPayload = { + ...retained.payload, + paneKey, + tabId: 'tab', + worktreeId: 'worktree', + connectionId: null, + stateStartedAt: now - 1, + receivedAt: now - 1, + workingMode: 'monitoring' + } + const sources = collectRuntimeWorktreeAgentSources({ + ...base, + hookSnapshots: [hook], + connectedPtyEvidence: { ...base.connectedPtyEvidence, ptyIds: new Set(['pty']) } + }) + expect(sources.get(paneKey)).toMatchObject({ updatedAt: now, workingMode: 'monitoring' }) + }) +}) diff --git a/src/main/runtime/runtime-worktree-agent-sources.ts b/src/main/runtime/runtime-worktree-agent-sources.ts new file mode 100644 index 00000000000..44ce7848960 --- /dev/null +++ b/src/main/runtime/runtime-worktree-agent-sources.ts @@ -0,0 +1,22 @@ +import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' +import { collectRuntimeWorktreePtyAgentSources } from './runtime-worktree-pty-agent-sources' +import { structuredRuntimeWorktreeAgentSources } from './runtime-worktree-structured-agent-rows' +import type { RuntimeWorktreeAgentSource } from './runtime-worktree-agent-source' + +/** One admitted roster for row and worktree-status projection. */ +export function collectRuntimeWorktreeAgentSources( + args: Parameters[0] & { + structuredSummaries: readonly AgentSessionStatusSummary[] + } +): ReadonlyMap { + const sources = new Map() + for (const source of collectRuntimeWorktreePtyAgentSources(args)) { + sources.set(source.paneKey, source) + } + for (const source of structuredRuntimeWorktreeAgentSources(args.structuredSummaries)) { + if (!sources.has(source.paneKey)) { + sources.set(source.paneKey, source) + } + } + return sources +} diff --git a/src/main/runtime/runtime-worktree-pty-agent-sources.ts b/src/main/runtime/runtime-worktree-pty-agent-sources.ts new file mode 100644 index 00000000000..04e8dffc04e --- /dev/null +++ b/src/main/runtime/runtime-worktree-pty-agent-sources.ts @@ -0,0 +1,121 @@ +import { + AGENT_STATUS_STALE_AFTER_MS, + pickParsedAgentStatusPayload, + type AgentStatusIpcPayload, + type ParsedAgentStatusPayload +} from '../../shared/agent-status-types' +import { terminalStatusPayloadMatchesHook } from '../../shared/agent-terminal-status-equivalence' +import { parseLegacyNumericPaneKey, parsePaneKey } from '../../shared/stable-pane-id' +import { isWslHookRelayConnectionId } from '../../shared/wsl-hook-relay-contract' +import type { RuntimeWorktreeAgentSource } from './runtime-worktree-agent-source' + +export type RuntimeAgentRowSnapshot = { + paneKey: string + ptyId: string + worktreeId?: string + tabId?: string + connectionId: string | null + payload: ParsedAgentStatusPayload + stateStartedAt: number + updatedAt: number +} + +export type ConnectedPtyEvidence = { + tabIds: ReadonlySet + paneKeys: ReadonlySet + ptyIds: ReadonlySet +} + +/** Reconcile terminal status, then admit rows using their execution-host evidence. */ +export function collectRuntimeWorktreePtyAgentSources(args: { + retainedSnapshots: Iterable + hookSnapshots: readonly AgentStatusIpcPayload[] + mirroredWorktreeIdByTabId: ReadonlyMap + connectedPtyEvidence: ConnectedPtyEvidence +}): RuntimeWorktreeAgentSource[] { + const rowSources = new Map< + string, + RuntimeWorktreeAgentSource & { payload: ParsedAgentStatusPayload } + >() + const now = Date.now() + for (const snapshot of args.retainedSnapshots) { + const { payload } = snapshot + rowSources.set(snapshot.paneKey, { + paneKey: snapshot.paneKey, + ptyId: snapshot.ptyId, + tabId: snapshot.tabId, + worktreeId: snapshot.worktreeId, + connectionId: snapshot.connectionId, + payload, + state: payload.state, + ...(payload.workingMode ? { workingMode: payload.workingMode } : {}), + agentType: payload.agentType ?? null, + prompt: payload.prompt, + lastAssistantMessage: payload.lastAssistantMessage ?? null, + toolName: payload.toolName ?? null, + toolInput: payload.toolInput ?? null, + interrupted: payload.interrupted ?? false, + stateStartedAt: snapshot.stateStartedAt, + updatedAt: snapshot.updatedAt + }) + } + for (const entry of args.hookSnapshots) { + if (entry.restoredUnconfirmed === true) { + continue + } + const existing = rowSources.get(entry.paneKey) + const hookPayload = pickParsedAgentStatusPayload(entry) + if (existing && existing.updatedAt > entry.receivedAt) { + if ( + entry.workingMode === 'monitoring' && + now - entry.receivedAt <= AGENT_STATUS_STALE_AFTER_MS && + terminalStatusPayloadMatchesHook(hookPayload, existing.payload) + ) { + existing.workingMode = 'monitoring' + if (existing.payload.workingMode === undefined) { + existing.payload = { ...existing.payload, workingMode: 'monitoring' } + } + } + continue + } + rowSources.set(entry.paneKey, { + paneKey: entry.paneKey, + ptyId: existing?.ptyId, + tabId: entry.tabId, + worktreeId: entry.worktreeId, + connectionId: entry.connectionId, + payload: hookPayload, + state: entry.state, + ...(entry.workingMode ? { workingMode: entry.workingMode } : {}), + agentType: entry.agentType ?? null, + prompt: entry.prompt, + lastAssistantMessage: entry.lastAssistantMessage ?? null, + toolName: entry.toolName ?? null, + toolInput: entry.toolInput ?? null, + interrupted: entry.interrupted ?? false, + stateStartedAt: entry.stateStartedAt, + updatedAt: entry.receivedAt + }) + } + const sources: RuntimeWorktreeAgentSource[] = [] + for (const source of rowSources.values()) { + const tabId = + source.tabId ?? + parsePaneKey(source.paneKey)?.tabId ?? + parseLegacyNumericPaneKey(source.paneKey)?.tabId + const mirroredWorktreeId = tabId ? args.mirroredWorktreeIdByTabId.get(tabId) : undefined + if ( + tabId !== undefined && + mirroredWorktreeId === undefined && + (source.connectionId === null || isWslHookRelayConnectionId(source.connectionId)) && + !args.connectedPtyEvidence.tabIds.has(tabId) && + !args.connectedPtyEvidence.paneKeys.has(source.paneKey) && + (source.ptyId === undefined || !args.connectedPtyEvidence.ptyIds.has(source.ptyId)) + ) { + continue + } + const worktreeId = mirroredWorktreeId ?? source.worktreeId + sources.push({ ...source, tabId, worktreeId }) + } + return sources +} diff --git a/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts b/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts new file mode 100644 index 00000000000..063ee0b36cf --- /dev/null +++ b/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts @@ -0,0 +1,157 @@ +import { collectRuntimeWorktreeAgentSources } from './runtime-worktree-agent-sources' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { StructuredAgentSessionStatusFeed } from '../native-chat/agent-session-wire/structured-agent-session-status-feed' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' +import type { RuntimeWorktreePsSummary } from '../../shared/runtime-types' +import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' + +/** + * The whole chain `worktree ps` walks: journal -> status feed -> agent rows -> worktree status. + * + * The feed's `published` map never retracts, so reading it as a roster reports every session the + * app has ever opened. A closed chat that was waiting on an approval is the sharp edge: deliberate + * close does not settle a pending prompt, so the retained summary stays `attention`, which maps to + * a `blocked` row and merges the worktree to `permission` for the 30-minute freshness window. + */ +const WORKTREE_ID = 'repo-1::/workspace/app' +const SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' +const IDENTITY = { + provider: 'codex', + threadId: 'thread-1', + turnId: 'turn-1', + ordinal: 0 +} as const + +let root: string +const journals = createTrackedJournalOpener() + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-structured-ps-liveness-')) +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +/** A session parked on an approval nobody answered — the state a deliberate close leaves behind. */ +async function awaitingApproval() { + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: WORKTREE_ID, + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: join(root, SESSION) + }) + await journal.appendItem( + { ...IDENTITY, ordinal: 1 }, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'rm the branch' }] }, + { fence: 1 } + ) + await journal.appendItem( + { ...IDENTITY, ordinal: 2 }, + { + kind: 'approval', + title: 'Run the command?', + detail: null, + options: [{ id: 'allow', label: 'Allow' }], + resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } + }, + { fence: 1 } + ) + const sessions = new Map([ + [ + SESSION, + { journal, params: { location: { workspaceId: WORKTREE_ID }, provider: 'codex' as const } } + ] + ]) + const feed = new StructuredAgentSessionStatusFeed({ + sessions, + getRecord: () => null, + now: () => Date.now() + }) + feed.publish(SESSION, journal) + return { feed, sessions } +} + +function worktreeFor( + feed: StructuredAgentSessionStatusFeed, + summaries = feed.liveSessionSummaries() +): RuntimeWorktreePsSummary { + const row = { + worktreeId: WORKTREE_ID, + status: 'inactive', + agents: [] + } as unknown as RuntimeWorktreePsSummary + attachRuntimeWorktreeAgentRows({ + summaries: new Map([[WORKTREE_ID, row]]), + pathIndex: { byPath: new Map(), byRealPath: new Map() } as never, + missingWorktreeIds: new Set(), + workingTerminalEvidenceByWorktreeId: new Map(), + rowSources: collectRuntimeWorktreeAgentSources({ + mirroredWorktreeIdByTabId: new Map(), + connectedPtyEvidence: { tabIds: new Set(), paneKeys: new Set(), ptyIds: new Set() }, + retainedSnapshots: [], + hookSnapshots: [], + structuredSummaries: summaries + }), + orchestrationByPaneKey: null, + getSummary: (map, _paths, _missing, id) => map.get(id) ?? null + }) + return row +} + +describe('worktree ps and a closed structured chat', () => { + it('reports the blocked row while the session is still held', async () => { + const { feed } = await awaitingApproval() + const row = worktreeFor(feed) + expect(row.agents).toHaveLength(1) + expect(row.agents[0]?.state).toBe('blocked') + expect(row.status).toBe('permission') + }) + + it('stops reporting it once eviction forgets the session', async () => { + const { feed, sessions } = await awaitingApproval() + // `forget-session`, the last eviction step, does exactly this and nothing to the feed. + sessions.delete(SESSION) + + const row = worktreeFor(feed) + expect(row.agents).toHaveLength(0) + expect(row.status).toBe('inactive') + }) + + it('keeps an aged host-held working state authoritative', async () => { + const { feed } = await awaitingApproval() + const aged = feed.liveSessionSummaries().map((summary) => ({ + ...summary, + hostExecutionOwned: true as const, + updatedAt: Date.now() - 30 * 60 * 1000 - 1, + status: 'working' as const + })) + const row = worktreeFor(feed, aged) + expect(row.agents).toHaveLength(1) + expect(row.agents[0]?.state).toBe('working') + expect(row.status).toBe('working') + expect(row.agents[0]?.updatedAt).toBe(aged[0]?.updatedAt) + }) + + it('keeps an aged host-held approval state authoritative', async () => { + const { feed } = await awaitingApproval() + const aged = feed.liveSessionSummaries().map((summary) => ({ + ...summary, + hostExecutionOwned: true as const, + updatedAt: Date.now() - 30 * 60 * 1000 - 1 + })) + const row = worktreeFor(feed, aged) + expect(row.agents).toHaveLength(1) + expect(row.agents[0]?.state).toBe('blocked') + expect(row.status).toBe('permission') + expect(row.agents[0]?.updatedAt).toBe(aged[0]?.updatedAt) + }) +}) diff --git a/src/main/runtime/runtime-worktree-structured-agent-rows.ts b/src/main/runtime/runtime-worktree-structured-agent-rows.ts new file mode 100644 index 00000000000..c4d075fbd9c --- /dev/null +++ b/src/main/runtime/runtime-worktree-structured-agent-rows.ts @@ -0,0 +1,48 @@ +import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' +import { + structuredAgentSessionPaneKey, + structuredAgentSessionStatusState, + structuredAgentSessionTabId +} from '../../shared/structured-agent-session-projection' +import type { RuntimeWorktreeAgentSource } from './runtime-worktree-agent-source' + +/** + * Row sources for the structured (non-PTY) sessions a host still holds. + * + * A structured session reaches none of the hook or retained snapshots every other row comes from, + * so `worktree ps` projects it from the host's status feed instead. The feed's retained + * projections are not a roster — the caller passes only sessions the host still holds. + */ +export function structuredRuntimeWorktreeAgentSources( + summaries: readonly AgentSessionStatusSummary[] +): RuntimeWorktreeAgentSource[] { + const sources: RuntimeWorktreeAgentSource[] = [] + for (const summary of summaries) { + // No turn has been persisted yet, so there is nothing to report - the same read the chat shows. + if (!summary.status) { + continue + } + const tabId = structuredAgentSessionTabId(summary.sessionId) + // The DERIVED pane key the renderer already publishes, never the orchestration bearer handle + // or the minted worker pane key: both of those are credentials. + sources.push({ + paneKey: structuredAgentSessionPaneKey(tabId, summary.sessionId), + tabId, + worktreeId: summary.workspaceId, + connectionId: null, + // The shared mapping the sidebar applies, so the CLI and the GUI cannot disagree about one + // session. No hook payload: nothing reads one off a structured row. + state: structuredAgentSessionStatusState(summary.status), + agentType: summary.agent, + prompt: summary.latestPrompt, + lastAssistantMessage: summary.lastAssistantMessage ?? null, + toolName: summary.toolName ?? null, + toolInput: summary.toolInput ?? null, + interrupted: false, + stateStartedAt: summary.updatedAt, + updatedAt: summary.updatedAt, + ...(summary.hostExecutionOwned ? { authority: 'structured-host' as const } : {}) + }) + } + return sources +} diff --git a/src/main/runtime/settled-worker-process-replacement.test.ts b/src/main/runtime/settled-worker-process-replacement.test.ts new file mode 100644 index 00000000000..2e575cd4546 --- /dev/null +++ b/src/main/runtime/settled-worker-process-replacement.test.ts @@ -0,0 +1,111 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' +import { OrchestrationDb } from './orchestration/db' + +const TAB = 'worker-tab' +const LEAF = '11111111-1111-4111-8111-111111111111' +const PANE = `${TAB}:${LEAF}` +const WORKSPACE = '/folder-workspace' +const LOCAL_HOST = JSON.stringify({ kind: 'local', hostId: 'local' }) +const SSH_HOST = JSON.stringify({ kind: 'ssh', targetId: 'remote-host' }) +let db: OrchestrationDb +let runtime: OrcaRuntimeService + +afterEach(() => { + db?.close() +}) + +function seedWorker(hostScope: string, settled = true) { + db = new OrchestrationDb(':memory:') + runtime = new OrcaRuntimeService(null) + runtime.setOrchestrationDb(db) + const started = db.createStartingWorkerDispatch({ + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + taskSpec: 'Ordinary pane after worker completion', + taskRunId: 'run_legacy_local', + startOptions: {} + }) + db.prepareStartingWorkerAuthority({ + dispatchId: started.dispatch.id, + handle: 'term_original', + paneKey: PANE, + processIncarnation: 'pty-original:inc-original', + hostScope, + worktreeId: WORKSPACE, + setupState: 'not_applicable', + effects: [], + terminalOwnership: 'created' + }) + db.markWorkerDispatchReady(started.dispatch.id) + if (settled) { + db.settleWorkerReport({ + taskId: started.task.id, + dispatchId: started.dispatch.id, + outcome: 'succeeded', + result: '{}' + }) + } + return { + dispatchId: started.dispatch.id, + task: db.getTask(started.task.id), + dispatch: db.getDispatchContextById(started.dispatch.id) + } +} + +function register(ptyId: string, incarnationId?: string, connectionId: string | null = null) { + runtime.registerPty(ptyId, WORKSPACE, connectionId, { + tabId: TAB, + leafId: LEAF, + ...(incarnationId ? { incarnationId } : {}) + }) +} + +describe('settled worker process replacement accounting', () => { + it.each([null, 'remote-host'])( + 'retains the replaced resource on owning host %s', + (connectionId) => { + const worker = seedWorker(connectionId ? SSH_HOST : LOCAL_HOST) + register('pty-resumed', 'inc-resumed', connectionId) + register('pty-resumed', 'inc-resumed', connectionId) + expect(db.getWorkerTerminalResourceByOwner(worker.dispatchId)).toMatchObject({ + release_state: 'retained', + retained_reason: 'identity_unproven', + process_incarnation: 'pty-original:inc-original' + }) + expect(db.getTask(worker.task!.id)).toEqual(worker.task) + expect(db.getDispatchContextById(worker.dispatchId)).toEqual(worker.dispatch) + expect(db.listWorkerTerminalResources({})).toEqual([ + expect.objectContaining({ dispatchId: worker.dispatchId, terminalState: 'retained' }) + ]) + } + ) + + it('keeps the original live resource unchanged across reattach', () => { + const worker = seedWorker(LOCAL_HOST) + const original = db.getWorkerTerminalResourceByOwner(worker.dispatchId) + register('pty-original', 'inc-original') + expect(db.getWorkerTerminalResourceByOwner(worker.dispatchId)).toEqual(original) + }) + + it('does not use missing incarnation evidence as proof of replacement', () => { + const worker = seedWorker(LOCAL_HOST) + const original = db.getWorkerTerminalResourceByOwner(worker.dispatchId) + register('pty-unverifiable') + expect(db.getWorkerTerminalResourceByOwner(worker.dispatchId)).toEqual(original) + }) + + it('does not change another execution host with the same pane and folder', () => { + const worker = seedWorker(SSH_HOST) + const original = db.getWorkerTerminalResourceByOwner(worker.dispatchId) + register('pty-resumed', 'inc-resumed') + expect(db.getWorkerTerminalResourceByOwner(worker.dispatchId)).toEqual(original) + }) + + it('does not change an active Dispatch resource', () => { + const worker = seedWorker(LOCAL_HOST, false) + const original = db.getWorkerTerminalResourceByOwner(worker.dispatchId) + register('pty-resumed', 'inc-resumed') + expect(db.getWorkerTerminalResourceByOwner(worker.dispatchId)).toEqual(original) + }) +}) diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index d9b3e59186a..0245f15054b 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -225,6 +225,8 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise + host?.publishBackgroundTaskState(sessionId, state), onEvent: (event) => { if (event.type !== 'ended' || !('cause' in event) || event.cause !== 'unexpected-exit') { return diff --git a/src/main/runtime/terminal-retirement-proof-emitted-frame.test.ts b/src/main/runtime/terminal-retirement-proof-emitted-frame.test.ts new file mode 100644 index 00000000000..c97d5429d18 --- /dev/null +++ b/src/main/runtime/terminal-retirement-proof-emitted-frame.test.ts @@ -0,0 +1,89 @@ +import { expect, it, vi } from 'vitest' +import type { + RuntimeMobileSessionTabsResult, + RuntimeMobileSessionTabsSnapshot +} from '../../shared/runtime-types' + +const { OrcaRuntimeService } = await import('./orca-runtime-test-mocks.spec') +await import('./orca-runtime-test-lifecycle.spec') +const { store, TEST_WORKTREE_ID } = await import('./orca-runtime-test-fixtures.spec') + +const retired = { + parentTabId: 'tab', + leafId: 'leaf', + ptyId: 'pty', + terminal: 'term_old', + incarnationId: 'inc' +} + +type RuntimeInternals = { + mobileSessionTabsByWorktree: Map + storeMobileSessionSnapshot: ( + worktreeId: string, + snapshot: RuntimeMobileSessionTabsSnapshot + ) => RuntimeMobileSessionTabsSnapshot +} + +function seedRuntimeWithStoredProof(): { + runtime: InstanceType + internals: RuntimeInternals +} { + const runtime = new OrcaRuntimeService(store) + runtime.setPtyController({ + spawn: vi.fn().mockResolvedValue({ id: 'pty-runtime-fallback' }), + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + }) + runtime.syncWindowGraph(0, { + tabs: [], + leaves: [], + mobileSessionTabs: [ + { + worktree: TEST_WORKTREE_ID, + publicationEpoch: 'headless:active-generation', + snapshotVersion: 7, + activeGroupId: null, + activeTabId: null, + activeTabType: null, + tabs: [] + } + ] + }) + const internals = runtime as unknown as RuntimeInternals + const stored = internals.mobileSessionTabsByWorktree.get(TEST_WORKTREE_ID)! + internals.storeMobileSessionSnapshot(TEST_WORKTREE_ID, { + ...stored, + snapshotVersion: stored.snapshotVersion + 1, + retiredTerminalSurfaces: [retired] + }) + return { runtime, internals } +} + +// Why: subscribers dedupe on (epoch, version), so a frame emitted at the stored version but +// built from the pre-store object would strand the proofs until an unrelated later bump. +it('emits the stored retirement proofs on the frame a runtime-owned create publishes', async () => { + const { runtime, internals } = seedRuntimeWithStoredProof() + const events: RuntimeMobileSessionTabsResult[] = [] + const unsubscribe = runtime.onMobileSessionTabsChanged( + (frame) => events.push(frame), + 'paired-client' + ) + + try { + await runtime.createMobileSessionTerminal(`id:${TEST_WORKTREE_ID}`, { + activate: false, + select: false, + navigation: 'caller', + clientNavigationId: 'paired-client' + }) + + const storedAfter = internals.mobileSessionTabsByWorktree.get(TEST_WORKTREE_ID)! + const emitted = events.at(-1)! + expect(storedAfter.retiredTerminalSurfaces).toEqual([retired]) + expect(emitted.snapshotVersion).toBe(storedAfter.snapshotVersion) + expect(emitted.retiredTerminalSurfaces).toEqual([retired]) + } finally { + unsubscribe() + } +}) diff --git a/src/main/runtime/terminal-retirement-proof-publication.test.ts b/src/main/runtime/terminal-retirement-proof-publication.test.ts new file mode 100644 index 00000000000..2ddb35a7e09 --- /dev/null +++ b/src/main/runtime/terminal-retirement-proof-publication.test.ts @@ -0,0 +1,31 @@ +import { expect, it } from 'vitest' +import { + createStaleTabCloseHarness, + WORKTREE_ID +} from './__fixtures__/orca-runtime-terminal-close-continuity-fixtures' + +it('keeps host retirement proof across a later renderer publication', async () => { + const harness = await createStaleTabCloseHarness({ headless: true }) + await harness.runtime.closeTerminalTab(harness.terminal.handle) + const before = await harness.runtime.listMobileSessionTabs(`id:${WORKTREE_ID}`) + expect(before.retiredTerminalSurfaces).toHaveLength(1) + + harness.runtime.syncWindowGraph(1, { + tabs: [], + leaves: [], + mobileSessionTabs: [ + { + worktree: WORKTREE_ID, + publicationEpoch: 'renderer:close-continuity', + snapshotVersion: 100, + activeGroupId: null, + activeTabId: null, + activeTabType: null, + tabs: [] + } + ] + }) + const after = await harness.runtime.listMobileSessionTabs(`id:${WORKTREE_ID}`) + expect(after.tabs).toEqual([]) + expect(after.retiredTerminalSurfaces).toEqual(before.retiredTerminalSurfaces) +}) diff --git a/src/main/runtime/workspace-session-membership-scaling.test.ts b/src/main/runtime/workspace-session-membership-scaling.test.ts new file mode 100644 index 00000000000..802245461b7 --- /dev/null +++ b/src/main/runtime/workspace-session-membership-scaling.test.ts @@ -0,0 +1,123 @@ +import { expect, it, vi } from 'vitest' +import type { TabGroup } from '../../shared/tab-types' +import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' +import { rebaseWorkspaceSessionTerminalMembership } from './workspace-session-terminal-membership-authority' + +it('rebases a large group without rescanning tab order for every recent tab', () => { + const ids = Array.from({ length: 1000 }, (_, index) => `tab-${index}`) + const session: WorkspaceSessionState = { + activeRepoId: 'repo', + activeWorktreeId: 'repo::/workspace', + activeTabId: null, + tabsByWorktree: { + 'repo::/workspace': ids.map((id, index) => ({ + id, + worktreeId: 'repo::/workspace', + ptyId: null, + title: id, + customTitle: null, + color: null, + sortOrder: index, + createdAt: 0 + })) + }, + terminalLayoutsByTabId: {}, + terminalTopologyRevisionByRepoId: { repo: 1 }, + tabGroups: { + 'repo::/workspace': [ + { + id: 'group', + worktreeId: 'repo::/workspace', + activeTabId: 'missing', + tabOrder: [...ids, 'missing'], + recentTabIds: [...ids, 'missing'] + } + ] + } + } + const includes = vi.spyOn(Array.prototype, 'includes') + let result: WorkspaceSessionState + let probes: number + try { + result = rebaseWorkspaceSessionTerminalMembership(session, session) + probes = includes.mock.calls.length + } finally { + includes.mockRestore() + } + expect(probes).toBeLessThan(10) + expect(result.tabGroups?.['repo::/workspace'][0]).toMatchObject({ + tabOrder: ids, + recentTabIds: ids, + activeTabId: ids[0] + }) +}) + +// Why: the rebased session is the host-authoritative one, so a tab id the host no +// longer has must not survive into it. `activeTabId` already failed closed; the +// filtered `recentTabIds` used to be dropped when it emptied, letting the `...group` +// spread put the unfiltered array back. +it('drops tab ids the host no longer has from group membership, failing closed', () => { + const buildSession = ( + activeTabId: string | null, + recentTabIds: string[] | undefined + ): WorkspaceSessionState => + ({ + activeRepoId: 'repo', + activeWorktreeId: 'repo::/workspace', + activeTabId: null, + tabsByWorktree: { + 'repo::/workspace': ['kept-a', 'kept-b'].map((id, index) => ({ + id, + worktreeId: 'repo::/workspace', + ptyId: null, + title: id, + customTitle: null, + color: null, + sortOrder: index, + createdAt: 0 + })) + }, + terminalLayoutsByTabId: {}, + terminalTopologyRevisionByRepoId: { repo: 1 }, + tabGroups: { + 'repo::/workspace': [ + { + id: 'group', + worktreeId: 'repo::/workspace', + activeTabId, + tabOrder: ['kept-a', 'closed-on-host', 'kept-b'], + ...(recentTabIds ? { recentTabIds } : {}) + } + ] + } + }) as WorkspaceSessionState + + const rebase = ( + activeTabId: string | null, + recentTabIds: string[] | undefined + ): TabGroup | undefined => { + const session = buildSession(activeTabId, recentTabIds) + return rebaseWorkspaceSessionTerminalMembership(session, session).tabGroups?.[ + 'repo::/workspace' + ][0] + } + + // Every recent id is stale: the array must empty, not revert to the stale one. + expect(rebase('kept-b', ['closed-on-host'])).toMatchObject({ + tabOrder: ['kept-a', 'kept-b'], + activeTabId: 'kept-b', + recentTabIds: [] + }) + // Mixed: only the host-known ids survive, in order. + expect(rebase('kept-a', ['closed-on-host', 'kept-b', 'closed-on-host', 'kept-a'])).toMatchObject({ + recentTabIds: ['kept-b', 'kept-a'] + }) + // A stale active tab falls back to the first surviving tab, never to the stale id. + expect(rebase('closed-on-host', ['kept-a'])).toMatchObject({ + activeTabId: 'kept-a', + recentTabIds: ['kept-a'] + }) + expect(rebase(null, undefined)).toMatchObject({ activeTabId: 'kept-a' }) + // A group that never carried recentTabIds must not gain the key. + expect(rebase('kept-a', undefined)).not.toHaveProperty('recentTabIds') +}) diff --git a/src/main/runtime/workspace-session-terminal-membership-authority.ts b/src/main/runtime/workspace-session-terminal-membership-authority.ts index 2869e3c813c..76cbcdf4dad 100644 --- a/src/main/runtime/workspace-session-terminal-membership-authority.ts +++ b/src/main/runtime/workspace-session-terminal-membership-authority.ts @@ -93,17 +93,18 @@ function rebaseTabGroups( if (tabOrder.length === 0) { return [] } + const tabIds = new Set(tabOrder) const activeTabId = - group.activeTabId && tabOrder.includes(group.activeTabId) - ? group.activeTabId - : (tabOrder[0] ?? null) - const recentTabIds = group.recentTabIds?.filter((tabId) => tabOrder.includes(tabId)) + group.activeTabId && tabIds.has(group.activeTabId) ? group.activeTabId : (tabOrder[0] ?? null) + const recentTabIds = group.recentTabIds?.filter((tabId) => tabIds.has(tabId)) return [ { ...group, tabOrder, activeTabId, - ...(recentTabIds && recentTabIds.length > 0 ? { recentTabIds } : {}) + // Why assigned even when it filters to empty: omitting the key lets `...group` + // re-introduce the unfiltered array, persisting ids for tabs the host dropped. + ...(group.recentTabIds ? { recentTabIds: recentTabIds ?? [] } : {}) } ] }) diff --git a/src/main/skills/agent-skill-selection.test.ts b/src/main/skills/agent-skill-selection.test.ts index 7dfa576211b..1b743374230 100644 --- a/src/main/skills/agent-skill-selection.test.ts +++ b/src/main/skills/agent-skill-selection.test.ts @@ -2,7 +2,8 @@ import { describe, expect, it } from 'vitest' import type { DiscoveredSkill } from '../../shared/skills' import { AGENT_SKILL_SELECTOR_AMBIGUOUS_CODE, - AGENT_SKILL_SELECTOR_NOT_FOUND_CODE + AGENT_SKILL_SELECTOR_NOT_FOUND_CODE, + AgentSkillSharingError } from '../../shared/agent-skill-sharing-contract' import { selectDiscoveredSkills } from './agent-skill-selection' @@ -50,3 +51,62 @@ describe('agent skill selection', () => { ).toThrow(expect.objectContaining({ code: AGENT_SKILL_SELECTOR_AMBIGUOUS_CODE })) }) }) + +it('indexes a batch of selectors without rescanning discovery', () => { + let reads = 0 + const skills = Array.from({ length: 1000 }, (_, index) => ({ + ...skill(`id-${index}`, `name-${index}`), + get id() { + reads++ + return `id-${index}` + } + })) + const selected = selectDiscoveredSkills( + skills, + skills.map((_, index) => `id-${index}`) + ) + expect(selected).toHaveLength(1000) + expect(selected[999]).toBe(skills[999]) + expect(reads).toBeLessThan(10000) +}) + +it('indexes only the requested selectors, not every discovered skill', () => { + let nameReads = 0 + const skills = Array.from({ length: 1000 }, (_, index) => ({ + ...skill(`id-${index}`, `name-${index}`), + get name() { + nameReads++ + return `name-${index}` + } + })) + expect(selectDiscoveredSkills(skills, ['id-900'])).toEqual([skills[900]]) + // One membership probe per discovered skill, plus reads for the single match's + // own bucket and the trailing collision check. Indexing every name would need + // three reads apiece. + expect(nameReads).toBeLessThanOrEqual(1_100) +}) + +// `matchingIds` is what the CLI prints so the user can disambiguate, so the +// index must report every match in discovery order, exactly like the old filter. +it('reports every ambiguous match in discovery order', () => { + let thrown: unknown + try { + selectDiscoveredSkills( + [skill('one', 'same'), skill('unrelated', 'other'), skill('two', 'same')], + ['same'] + ) + } catch (error) { + thrown = error + } + expect(thrown).toBeInstanceOf(AgentSkillSharingError) + const error = thrown as AgentSkillSharingError + expect(error.code).toBe(AGENT_SKILL_SELECTOR_AMBIGUOUS_CODE) + expect(error.data).toEqual({ selector: 'same', matchingIds: ['one', 'two'] }) +}) + +it('retains first duplicate ID authority and exact ID precedence over names', () => { + const first = skill('id', 'first') + expect( + selectDiscoveredSkills([first, skill('id', 'second'), skill('other', 'id')], ['id']) + ).toEqual([first]) +}) diff --git a/src/main/skills/agent-skill-selection.ts b/src/main/skills/agent-skill-selection.ts index 23fb5581cab..e38dd3767d2 100644 --- a/src/main/skills/agent-skill-selection.ts +++ b/src/main/skills/agent-skill-selection.ts @@ -10,13 +10,35 @@ export function selectDiscoveredSkills( selectors: readonly string[] ): DiscoveredSkill[] { const selected = new Map() + // Indexed only for the selectors actually asked for: a share request carries at + // most 512 of them while discovery can return every skill on the machine, and + // indexing the whole set costs more than the scans it replaces for the + // one-or-two-selector requests agents actually send. + const requested = new Set(selectors) + const byId = new Map() + const discoveredByName = new Map() + for (const skill of skills) { + // First writer wins, matching the `find` this replaces. + if (requested.has(skill.id) && !byId.has(skill.id)) { + byId.set(skill.id, skill) + } + if (!requested.has(skill.name)) { + continue + } + const named = discoveredByName.get(skill.name) + if (named) { + named.push(skill) + } else { + discoveredByName.set(skill.name, [skill]) + } + } for (const selector of selectors) { - const exactId = skills.find((skill) => skill.id === selector) + const exactId = byId.get(selector) if (exactId) { selected.set(exactId.id, exactId) continue } - const named = skills.filter((skill) => skill.name === selector) + const named = discoveredByName.get(selector) ?? [] if (named.length === 0) { throw new AgentSkillSharingError( AGENT_SKILL_SELECTOR_NOT_FOUND_CODE, @@ -36,7 +58,12 @@ export function selectDiscoveredSkills( const values = [...selected.values()] const byName = new Map() for (const skill of values) { - byName.set(skill.name, [...(byName.get(skill.name) ?? []), skill]) + const named = byName.get(skill.name) + if (named) { + named.push(skill) + } else { + byName.set(skill.name, [skill]) + } } const collision = [...byName.entries()].find(([, named]) => named.length > 1) if (collision) { diff --git a/src/main/skills/discovery.ts b/src/main/skills/discovery.ts index a93968177a9..385ae99ca33 100644 --- a/src/main/skills/discovery.ts +++ b/src/main/skills/discovery.ts @@ -10,7 +10,8 @@ import type { } from '../../shared/skills' import { buildSkillDiscoverySources, - compareSkills, + sortDiscoveredSkills, + sortSkillDiscoverySources, sourceKindForSkill, sourceLabelForSkill, stablePathId, @@ -292,7 +293,7 @@ export async function discoverSkills(args: { mergeScannedSkill(seen, skill) } } - const skills = Array.from(seen.values()).sort(compareSkills) + const skills = sortDiscoveredSkills(Array.from(seen.values())) // Why: root *ids* — a repo/plugin id is already a hash, while its label carries // the repo or plugin name and its path carries the user's directory names. A // fully cached scan did no filesystem work, so it stays silent rather than @@ -309,9 +310,7 @@ export async function discoverSkills(args: { } return { skills, - sources: sources.sort((a, b) => - a.label.localeCompare(b.label, undefined, { sensitivity: 'base' }) - ), + sources: sortSkillDiscoverySources(sources), scannedAt: Date.now() } } diff --git a/src/main/skills/skill-cloud-grant-installation.test.ts b/src/main/skills/skill-cloud-grant-installation.test.ts index 11d15a30689..5355e6f4869 100644 --- a/src/main/skills/skill-cloud-grant-installation.test.ts +++ b/src/main/skills/skill-cloud-grant-installation.test.ts @@ -133,3 +133,89 @@ describe('installSkillCloudGrant', () => { ) }) }) + +// The failure report is keyed by user-visible skill IDs, so switching the +// membership test from `includes` to a Set must not change which entries appear, +// how often, or in what order. +it('reports selected manifest entries in manifest order, duplicates and all', async () => { + const skills = ['b', 'a', 'dupe', 'dupe', 'unselected'].map((id) => ({ + id, + name: `name-${id}`, + digest: 'a'.repeat(64), + files: [] + })) + const bundleGrant = { + ...grant, + version: { ...grant.version, manifest: { skills, bundleDigest: 'c'.repeat(64) } } + } as unknown as SkillCloudDownloadGrant + const runtime = { + installSharedSkillBundleRequest: vi + .fn() + .mockRejectedValue(new Error('skill-install-filesystem-failed')) + } as unknown as OrcaRuntimeService + const result = await installSkillBundleCloudGrant(runtime, bundleGrant, { + operationId: 'op', + // Repeated and unknown selections must be inert, exactly as with `includes`. + selectedSkillIds: ['dupe', 'a', 'a', 'b', 'never-in-manifest'], + destination: { scope: 'global' } + }) + expect(result.status).toBe('ok') + if (result.status === 'ok') { + expect(result.value.skills.map((skill) => skill.skillId)).toEqual(['b', 'a', 'dupe', 'dupe']) + } +}) + +it('reports no skills when nothing was selected', async () => { + const skills = [{ id: 'a', name: 'a', digest: 'a'.repeat(64), files: [] }] + const bundleGrant = { + ...grant, + version: { ...grant.version, manifest: { skills, bundleDigest: 'c'.repeat(64) } } + } as unknown as SkillCloudDownloadGrant + const runtime = { + installSharedSkillBundleRequest: vi.fn().mockRejectedValue(new Error('skill-install-cancelled')) + } as unknown as OrcaRuntimeService + const result = await installSkillBundleCloudGrant(runtime, bundleGrant, { + operationId: 'op', + selectedSkillIds: [], + destination: { scope: 'global' } + }) + expect(result.status).toBe('ok') + if (result.status === 'ok') { + expect(result.value.skills).toEqual([]) + } +}) + +it.each(['skill-install-cancelled', 'skill-install-filesystem-failed'])( + 'indexes selected IDs when reporting %s', + async (code) => { + let reads = 0 + const ids = Array.from({ length: 1000 }, (_, index) => `skill-${index}`) + const selectedSkillIds = new Proxy(ids, { + get(target, key, receiver) { + if (typeof key === 'string' && /^\d+$/.test(key)) { + reads += 1 + } + return Reflect.get(target, key, receiver) + } + }) + const skills = ids.map((id) => ({ id, name: id, digest: 'a'.repeat(64), files: [] })) + const bundleGrant = { + ...grant, + version: { ...grant.version, manifest: { skills, bundleDigest: 'c'.repeat(64) } } + } as unknown as SkillCloudDownloadGrant + const runtime = { + installSharedSkillBundleRequest: vi.fn().mockRejectedValue(new Error(code)) + } as unknown as OrcaRuntimeService + const result = await installSkillBundleCloudGrant(runtime, bundleGrant, { + operationId: 'op', + selectedSkillIds, + destination: { scope: 'global' } + }) + expect(result.status).toBe('ok') + if (result.status === 'ok') { + expect(result.value.skills.map((skill) => skill.skillId)).toEqual(ids) + expect(result.value.status).toBe(code.includes('cancelled') ? 'cancelled' : 'failed') + } + expect(reads).toBeLessThanOrEqual(2000) + } +) diff --git a/src/main/skills/skill-cloud-grant-installation.ts b/src/main/skills/skill-cloud-grant-installation.ts index add69270c5d..e60c8712820 100644 --- a/src/main/skills/skill-cloud-grant-installation.ts +++ b/src/main/skills/skill-cloud-grant-installation.ts @@ -49,6 +49,7 @@ function bundleFailureResult( if (!('skills' in manifest)) { throw new Error('skill-bundle-cloud-manifest-required') } + const selectedSkillIds = new Set(request.selectedSkillIds) return { operationId: request.operationId, packageId: request.package.packageId, @@ -56,7 +57,7 @@ function bundleFailureResult( bundleDigest: request.package.bundleDigest, status: failure.category === 'cancelled' ? 'cancelled' : 'failed', skills: manifest.skills - .filter((skill) => request.selectedSkillIds.includes(skill.id)) + .filter((skill) => selectedSkillIds.has(skill.id)) .map((skill) => ({ skillId: skill.id, name: skill.name, diff --git a/src/main/skills/skill-discovery-order.test.ts b/src/main/skills/skill-discovery-order.test.ts new file mode 100644 index 00000000000..18500bdf00e --- /dev/null +++ b/src/main/skills/skill-discovery-order.test.ts @@ -0,0 +1,153 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { DiscoveredSkill, SkillDiscoverySource } from '../../shared/skills' +import { sortDiscoveredSkills, sortSkillDiscoverySources } from './skill-discovery-sources' + +// Scripts and case/accent/numeric shapes whose collation differs between locales +// and ICU builds, so a reused collator that drifted from `localeCompare` shows up. +const COLLATION_CORPUS = [ + '', + ' ', + 'a', + 'A', + 'éclair', + 'Eclair', + 'ÉCLAIR', + 'Ångström', + 'Angstrom', + 'İstanbul', + 'Istanbul', + 'ıstanbul', + 'straße', + 'strasse', + 'STRASSE', + 'ẞ', + '中文', + '日本語', + 'にほんご', + '한국어', + 'item2', + 'item10', + 'item01', + 'ITEM2', + '10', + '2', + 'ñ', + 'n', + 'œ', + 'oe', + 'æ', + 'привет', + 'ПРИВЕТ', + 'skill-a', + 'skill_a', + 'skill a', + 'co-op', + 'coop', + 'zebra', + 'zebra' +] + +function skill(index: number): DiscoveredSkill { + return { + id: String(index), + name: ['éclair', 'Eclair', 'item2', 'item10', 'Ångström', 'zebra', 'İstanbul'][index % 7], + description: null, + providers: ['codex'], + sourceKind: 'home', + sourceLabel: ['Home', 'hôme', 'Repo', 'repo'][index % 4], + rootPath: '/skills', + directoryPath: '/skills/example', + skillFilePath: `/skills/${index % 13}/SKILL.md`, + installed: true, + updatedAt: null + } +} + +// Preserve the original comparator as the ordering and operation-count oracle. +function compareOriginal(a: DiscoveredSkill, b: DiscoveredSkill): number { + return ( + a.name.localeCompare(b.name, undefined, { sensitivity: 'base' }) || + a.sourceLabel.localeCompare(b.sourceLabel, undefined, { sensitivity: 'base' }) || + a.skillFilePath.localeCompare(b.skillFilePath) + ) +} + +function discoverySource(index: number): SkillDiscoverySource { + return { + id: String(index), + label: COLLATION_CORPUS[index % COLLATION_CORPUS.length], + path: `/roots/${index}`, + sourceKind: 'home', + providers: ['codex'], + owner: null, + exists: true + } +} + +afterEach(() => vi.restoreAllMocks()) + +describe('discovered skill ordering', () => { + it('preserves name, source, path and stable ties with one collator per sort', () => { + const skills = Array.from({ length: 2_000 }, (_, index) => skill(index)) + const localeCompare = vi.spyOn(String.prototype, 'localeCompare') + const expected = [...skills].sort(compareOriginal) + const optionedCalls = (): number => + localeCompare.mock.calls.filter((args) => args[2] !== undefined).length + expect(optionedCalls()).toBeGreaterThan(10_000) + localeCompare.mockClear() + const NativeCollator = Intl.Collator + const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + return new NativeCollator(locales, options) + }) + + expect(sortDiscoveredSkills(skills)).toBe(skills) + expect(skills.map(({ id }) => id)).toEqual(expected.map(({ id }) => id)) + expect(optionedCalls()).toBe(0) + expect(construct).toHaveBeenCalledExactlyOnceWith(undefined, { sensitivity: 'base' }) + sortDiscoveredSkills([...skills]) + expect(construct).toHaveBeenCalledTimes(2) + }) + + it('matches the per-call comparator across scripts, case, accents and numerals', () => { + const skills = COLLATION_CORPUS.flatMap((name, index) => + COLLATION_CORPUS.map((sourceLabel, inner) => ({ + ...skill(index), + id: `${index}-${inner}`, + name, + sourceLabel + })) + ) + expect(skills.length).toBe(COLLATION_CORPUS.length ** 2) + const expected = [...skills].sort(compareOriginal) + expect(sortDiscoveredSkills([...skills]).map(({ id }) => id)).toEqual( + expected.map(({ id }) => id) + ) + }) + + it('sorts discovery sources like the per-call label comparator', () => { + const sources = Array.from({ length: 400 }, (_, index) => discoverySource(index)) + const expected = [...sources].sort((a, b) => + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Parity oracle. + a.label.localeCompare(b.label, undefined, { sensitivity: 'base' }) + ) + const NativeCollator = Intl.Collator + const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + return new NativeCollator(locales, options) + }) + expect(sortSkillDiscoverySources(sources).map(({ id }) => id)).toEqual( + expected.map(({ id }) => id) + ) + expect(construct).toHaveBeenCalledExactlyOnceWith(undefined, { sensitivity: 'base' }) + }) + + it('does no comparison setup for empty or singleton discovery results', () => { + const construct = vi.spyOn(Intl, 'Collator') + for (const skills of [[], [skill(0)]]) { + expect(sortDiscoveredSkills(skills)).toBe(skills) + } + for (const sources of [[], [discoverySource(0)]]) { + expect(sortSkillDiscoverySources(sources)).toBe(sources) + } + expect(construct).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/skills/skill-discovery-sources.ts b/src/main/skills/skill-discovery-sources.ts index 9d9781c3687..09bf8e208c8 100644 --- a/src/main/skills/skill-discovery-sources.ts +++ b/src/main/skills/skill-discovery-sources.ts @@ -47,14 +47,27 @@ export function sourceLabelForSkill(root: SkillScanRoot, sourceKind: SkillSource return sourceKind === 'bundled' ? `${root.label} bundled` : root.label } -export function compareSkills(a: DiscoveredSkill, b: DiscoveredSkill): number { - return ( - a.name.localeCompare(b.name, undefined, { sensitivity: 'base' }) || - a.sourceLabel.localeCompare(b.sourceLabel, undefined, { sensitivity: 'base' }) || - a.skillFilePath.localeCompare(b.skillFilePath) +export function sortDiscoveredSkills(skills: DiscoveredSkill[]): DiscoveredSkill[] { + if (skills.length < 2) { + return skills + } + const compare = new Intl.Collator(undefined, { sensitivity: 'base' }).compare + return skills.sort( + (a, b) => + compare(a.name, b.name) || + compare(a.sourceLabel, b.sourceLabel) || + a.skillFilePath.localeCompare(b.skillFilePath) ) } +export function sortSkillDiscoverySources(sources: SkillDiscoverySource[]): SkillDiscoverySource[] { + if (sources.length < 2) { + return sources + } + const compare = new Intl.Collator(undefined, { sensitivity: 'base' }).compare + return sources.sort((a, b) => compare(a.label, b.label)) +} + function source( id: string, label: string, diff --git a/src/main/skills/skill-discovery-wsl.test.ts b/src/main/skills/skill-discovery-wsl.test.ts index 2bcab34355d..93640698dd0 100644 --- a/src/main/skills/skill-discovery-wsl.test.ts +++ b/src/main/skills/skill-discovery-wsl.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import type { SkillScanRoot } from './skill-discovery-sources' import { buildWslSkillDiscoveryCommand, parseWslSkillDiscoveryOutput } from './skill-discovery-wsl' @@ -88,3 +88,29 @@ describe('WSL skill discovery', () => { ) }) }) + +it('reuses one source collator while preserving locale, lexical numbers, and stable ties', () => { + const labels = Array.from( + { length: 200 }, + (_, index) => + ['éclair', 'Eclair', 'item2', 'item10', 'Ångström', 'zebra', 'İstanbul'][index % 7] + ) + const roots = labels.map((label, index) => ({ ...homeRoot, id: String(index), label })) + const expected = [...roots].sort((a, b) => + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Preserve the old comparator as the parity oracle. + a.label.localeCompare(b.label, undefined, { sensitivity: 'base' }) + ) + const NativeCollator = Intl.Collator + const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + return new NativeCollator(locales, options) + }) + const localeCompare = vi.spyOn(String.prototype, 'localeCompare') + try { + const result = parseWslSkillDiscoveryOutput('', roots, 42) + expect(result.sources.map((source) => source.id)).toEqual(expected.map((root) => root.id)) + expect(construct).toHaveBeenCalledExactlyOnceWith(undefined, { sensitivity: 'base' }) + expect(localeCompare).not.toHaveBeenCalled() + } finally { + vi.restoreAllMocks() + } +}) diff --git a/src/main/skills/skill-discovery-wsl.ts b/src/main/skills/skill-discovery-wsl.ts index c9bbcbb96cb..eae829a893f 100644 --- a/src/main/skills/skill-discovery-wsl.ts +++ b/src/main/skills/skill-discovery-wsl.ts @@ -9,7 +9,8 @@ import { quoteBashString } from '../wsl-bash-command' import { runWslProcess } from '../wsl/wsl-runner' import { buildSkillDiscoverySources, - compareSkills, + sortDiscoveredSkills, + sortSkillDiscoverySources, sourceKindForSkill, sourceLabelForSkill, stablePathId, @@ -162,10 +163,8 @@ export function parseWslSkillDiscoveryOutput( } }) return { - skills: [...skillsByCanonicalPath.values()].sort(compareSkills), - sources: sources.sort((a, b) => - a.label.localeCompare(b.label, undefined, { sensitivity: 'base' }) - ), + skills: sortDiscoveredSkills([...skillsByCanonicalPath.values()]), + sources: sortSkillDiscoverySources(sources), scannedAt } } diff --git a/src/main/skills/skill-update-convergence.test.ts b/src/main/skills/skill-update-convergence.test.ts index 3e8d2633a06..7f30a799282 100644 --- a/src/main/skills/skill-update-convergence.test.ts +++ b/src/main/skills/skill-update-convergence.test.ts @@ -169,4 +169,34 @@ describe('convergableSkillNames', () => { ) expect([...result]).toEqual(['orca-cli']) }) + // A skill directory can legitimately be named `constructor`, and lock names come + // straight off disk, so the snapshot lookup must not walk Object.prototype. + it('keeps a skill named after an Object prototype key eligible instead of throwing', () => { + for (const name of ['constructor', 'toString', 'hasOwnProperty', '__proto__']) { + expect([ + ...convergableSkillNames([placement(name, 1)], new Map([[name, '091d9bcc']]), {}) + ]).toEqual([name]) + } + }) + + it('indexes placements once across many independent locked skills', () => { + let nameReads = 0 + const installations = Array.from({ length: 1000 }, (_, index) => ({ + ...placement(`skill-${index}`, index % 2 ? 2 : 1), + get name() { + nameReads++ + return `skill-${index}` + } + })) + const locks = new Map(installations.map((entry) => [entry.name, STUB.gitTreeSha!])) + const snapshots = Object.fromEntries([...locks.keys()].map((name) => [name, [PRE_STUB, STUB]])) + nameReads = 0 + expect([...convergableSkillNames(installations, locks, snapshots)]).toEqual( + [...locks.keys()].filter((_, index) => index % 2 === 1) + ) + expect(nameReads).toBeLessThanOrEqual(1000) + nameReads = 0 + expect([...convergableSkillNames(installations, new Map(), snapshots)]).toEqual([]) + expect(nameReads).toBe(0) + }) }) diff --git a/src/main/skills/skill-update-convergence.ts b/src/main/skills/skill-update-convergence.ts index c318e89d8b8..e87200662ff 100644 --- a/src/main/skills/skill-update-convergence.ts +++ b/src/main/skills/skill-update-convergence.ts @@ -28,22 +28,39 @@ export function convergableSkillNames( knownSnapshots: Readonly> ): ReadonlySet { const convergable = new Set(globalSkillLocks.keys()) + if (convergable.size === 0) { + return convergable + } + const observableByName = new Map() + for (const entry of installations) { + const name = entry.name + if ( + !globalSkillLocks.has(name) || + !SUPPORTED_GLOBAL_SKILL_TOPOLOGIES.has(entry.topology) || + !entry.observedPackageDigest + ) { + continue + } + const entries = observableByName.get(name) + if (entries) { + entries.push(entry) + } else { + observableByName.set(name, [entry]) + } + } for (const [name, lockHash] of globalSkillLocks) { // Why: judged only over the placements the command writes, like eligibility // itself. A plugin-cache or repo copy is never the command's to converge, so // it must neither gate the name nor rescue it — an unidentifiable cache copy // (or one parked at the lock's own revision) would otherwise defeat the gate // and re-arm the unwinnable update. - const observable = installations.filter( - (entry) => - entry.name === name && - SUPPORTED_GLOBAL_SKILL_TOPOLOGIES.has(entry.topology) && - entry.observedPackageDigest - ) + const observable = observableByName.get(name) ?? [] if (observable.length === 0) { continue } - const revisions = knownSnapshots[name] ?? [] + // `Object.hasOwn`: names come from an on-disk lock file, so a skill called + // `constructor` would otherwise read a function off the prototype and throw. + const revisions = (Object.hasOwn(knownSnapshots, name) && knownSnapshots[name]) || [] // Why: the revision each placement resolved to during observation, not a fresh // lookup by whole-folder digest. Identity tolerates files the manifest never // listed, so a folder holding an agent CLI's sidecar digests to nothing any diff --git a/src/main/startup/main-process-runtime-service.ts b/src/main/startup/main-process-runtime-service.ts index a684e951bc3..1b7f69213ad 100644 --- a/src/main/startup/main-process-runtime-service.ts +++ b/src/main/startup/main-process-runtime-service.ts @@ -129,7 +129,6 @@ export function initializeMainProcessRuntime(): OrcaRuntimeService { agentHookServer.subscribeEnrichedStatus((enriched) => recordObservedAgentStatusPaneIdentity(observedPaneIdentities, enriched.paneKey, runtime) ) - runtime.prepareLegacyWorkerTerminalRecovery() // Why before anything can attach: a client host that reattaches to a restarted runtime is only // handed its pages back if the runtime found them first. runtime.rehydrateClientHostedBrowserPages() diff --git a/src/main/usage/highest-usage-key.test.ts b/src/main/usage/highest-usage-key.test.ts new file mode 100644 index 00000000000..5fea5c33182 --- /dev/null +++ b/src/main/usage/highest-usage-key.test.ts @@ -0,0 +1,57 @@ +import { expect, it } from 'vitest' +import { highestUsageKey } from './highest-usage-key' + +it('reads each total once instead of sorting all projects for one winner', () => { + let reads = 0 + const entries = Array.from({ length: 2000 }, (_, index): [string, number] => { + const pair: [string, number] = [String(index), (index * 173) % 2000] + Object.defineProperty(pair, '1', { + get() { + reads += 1 + return (index * 173) % 2000 + } + }) + return pair + }) + const totals = new Map(entries) + Object.defineProperty(totals, Symbol.iterator, { value: () => entries[Symbol.iterator]() }) + reads = 0 + const expected = [...totals].sort((left, right) => right[1] - left[1])[0][0] + expect(reads).toBeGreaterThan(10000) + reads = 0 + expect(highestUsageKey(totals)).toBe(expected) + expect(reads).toBe(2000) +}) + +it('breaks ties on the first-inserted key, including after a later re-set', () => { + expect( + highestUsageKey( + new Map([ + ['zebra', 10], + ['alpha', 10], + ['mid', 10] + ]) + ) + ).toBe('zebra') + const reset = new Map() + reset.set('first', 1) + reset.set('second', 5) + reset.set('first', 5) + expect(highestUsageKey(reset)).toBe('first') +}) + +it('preserves empty, first-tie, negative and nonfinite ordering behavior', () => { + for (const values of [ + [], + [1, 1, 0], + [-4, -2], + [1, Number.NaN, 2], + [Infinity, Infinity, 1], + [-Infinity, -Infinity] + ]) { + const totals = new Map(values.map((value, index) => [String(index), value])) + expect(highestUsageKey(totals)).toBe( + [...totals].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null + ) + } +}) diff --git a/src/main/usage/highest-usage-key.ts b/src/main/usage/highest-usage-key.ts new file mode 100644 index 00000000000..5844f51dff2 --- /dev/null +++ b/src/main/usage/highest-usage-key.ts @@ -0,0 +1,18 @@ +// Replaces a stable descending sort, so ties must resolve to the first-inserted +// key — hence the strict `>` below rather than `>=`. +export function highestUsageKey(totals: ReadonlyMap): string | null { + let bestKey: string | null = null + let bestTotal = Number.NEGATIVE_INFINITY + for (const [key, total] of totals) { + if (Number.isNaN(total)) { + // NaN makes the old comparator inconsistent; defer to it verbatim so a + // corrupt total cannot change which key the summary reports. + return [...totals].sort((left, right) => right[1] - left[1])[0]?.[0] ?? null + } + if (bestKey === null || total > bestTotal) { + bestKey = key + bestTotal = total + } + } + return bestKey +} diff --git a/src/main/usage/usage-breakdown-scaling.test.ts b/src/main/usage/usage-breakdown-scaling.test.ts new file mode 100644 index 00000000000..2412d5d097d --- /dev/null +++ b/src/main/usage/usage-breakdown-scaling.test.ts @@ -0,0 +1,130 @@ +import { expect, it } from 'vitest' +import { createUsageEventAggregation } from './usage-event-aggregation' +import type { UsageAttributedEventFields } from './usage-rollup-records' + +type Event = UsageAttributedEventFields & { cost: number } +const aggregation = createUsageEventAggregation({ + metric: { + empty: () => ({ cost: 0 }), + fromEvent: (event) => ({ cost: event.cost }), + fold: (target, source) => { + target.cost += source.cost + } + }, + cloneSessionForMerge: (session) => structuredClone(session) +}) + +function events(count: number): Event[] { + return Array.from({ length: count }, (_, index) => ({ + sessionId: 'session', + timestamp: '2026-09-07T00:00:00Z', + day: '2026-09-07', + model: `model-${index % 5}`, + projectKey: `path-${index}`, + projectLabel: `Path ${index}`, + repoId: null, + worktreeId: null, + inputTokens: 1, + cachedInputTokens: 0, + outputTokens: 2, + reasoningOutputTokens: 0, + totalTokens: 3, + cost: 0.5 + })) +} + +it('folds events without rescanning accumulated location breakdowns', () => { + let reads = 0 + const input = events(1000) + input.forEach((event, index) => + Object.defineProperty(event, 'projectKey', { + get() { + reads += 1 + return `path-${index}` + } + }) + ) + const result = aggregation.aggregate([...input, ...input]) + expect(reads).toBeLessThan(30000) + const session = result.sessions[0] + expect(session.totalTokens).toBe(6000) + expect(session.cost).toBe(1000) + expect(session.locationBreakdown).toHaveLength(1000) + expect(session.locationBreakdown.every((entry) => entry.eventCount === 2)).toBe(true) + expect(session.modelBreakdown).toHaveLength(5) + expect(result.dailyAggregates).toHaveLength(1000) +}) + +it('merges rollups using first-match indexes without mutating source breakdowns', () => { + const source = aggregation.aggregate(events(1000)).sessions[0] + const existing = structuredClone(source) + let reads = 0 + for (const entry of [...existing.locationBreakdown, ...existing.locationModelBreakdown]) { + const key = entry.locationKey + Object.defineProperty(entry, 'locationKey', { + get() { + reads += 1 + return key + } + }) + } + const target = new Map([['session', existing]]) + aggregation.mergeSessions(target, [source, source]) + expect(reads).toBeLessThan(10000) + expect(existing.totalTokens).toBe(9000) + expect(existing.cost).toBe(1500) + expect(source.totalTokens).toBe(3000) + expect(existing.locationBreakdown.every((entry) => entry.eventCount === 3)).toBe(true) +}) + +it('preserves first-wins duplicate rows and exact location/model tuple identity', () => { + const source = aggregation.aggregate(events(1)).sessions[0] + const existing = structuredClone(source) + existing.locationBreakdown.push({ ...existing.locationBreakdown[0], eventCount: 42 }) + aggregation.mergeSessions(new Map([['session', existing]]), [source]) + expect(existing.locationBreakdown.map((entry) => entry.eventCount)).toEqual([2, 42]) + const input = events(2) + input[0].projectKey = 'a::b' + input[0].model = 'c' + input[1].projectKey = 'a' + input[1].model = 'b::c' + expect(aggregation.aggregate(input).sessions[0].locationModelBreakdown).toHaveLength(2) +}) + +// Quotes and backslashes are the shapes a plain `::` separator would still collapse. +it('keeps location/model tuples distinct under quote and backslash keys', () => { + const input = events(4) + input[0].projectKey = 'a"' + input[0].model = 'b' + input[1].projectKey = 'a' + input[1].model = '"b' + input[2].projectKey = 'a\\' + input[2].model = 'b' + input[3].projectKey = 'a' + input[3].model = '\\b' + const session = aggregation.aggregate(input).sessions[0] + expect(session.locationModelBreakdown).toHaveLength(4) + expect(session.locationModelBreakdown.every((entry) => entry.eventCount === 1)).toBe(true) +}) + +// The merge index must learn the rows it appends, or a second source carrying the same +// location/model would append a duplicate row instead of folding into the first. +it('folds later sources into rows the merge itself appended', () => { + const [first] = events(1) + first.projectKey = 'new-location' + first.projectLabel = 'New location' + first.model = 'new-model' + const incoming = aggregation.aggregate([first]).sessions[0] + const existingOnly = events(1) + existingOnly[0].projectKey = 'other' + const existing = aggregation.aggregate(existingOnly).sessions[0] + aggregation.mergeSessions(new Map([['session', existing]]), [incoming, structuredClone(incoming)]) + const rows = existing.locationBreakdown.filter((entry) => entry.locationKey === 'new-location') + expect(rows.map((entry) => entry.eventCount)).toEqual([2]) + const models = existing.modelBreakdown.filter((entry) => entry.modelKey === 'new-model') + expect(models.map((entry) => entry.eventCount)).toEqual([2]) + const tuples = existing.locationModelBreakdown.filter( + (entry) => entry.locationKey === 'new-location' && entry.modelKey === 'new-model' + ) + expect(tuples.map((entry) => entry.eventCount)).toEqual([2]) +}) diff --git a/src/main/usage/usage-event-aggregation.ts b/src/main/usage/usage-event-aggregation.ts index ca3eee17544..d58a854fb37 100644 --- a/src/main/usage/usage-event-aggregation.ts +++ b/src/main/usage/usage-event-aggregation.ts @@ -2,6 +2,11 @@ * Folds attributed usage events into per-session and per-day rollups. Shared by every * event-based usage provider so a token-accounting fix lands in all of them at once. */ +import { + indexUsageSessionBreakdowns, + usageLocationModelKey, + type UsageSessionBreakdownIndex +} from './usage-session-breakdown-index' import { mergeUsageDailyAggregates, mergeUsageSessions } from './usage-rollup-merge' import { usageDailyAggregateKey, @@ -71,9 +76,10 @@ export function createUsageEventAggregation< function foldLocation( target: UsageLocationBreakdown[], event: TEvent, - eventMetric: TMetric + eventMetric: TMetric, + index: UsageSessionBreakdownIndex['locations'] ): void { - const existing = target.find((entry) => entry.locationKey === event.projectKey) ?? null + const existing = index.get(event.projectKey) if (existing) { existing.eventCount++ existing.inputTokens += event.inputTokens @@ -85,7 +91,7 @@ export function createUsageEventAggregation< return } - target.push({ + const entry: UsageLocationBreakdown = { locationKey: event.projectKey, projectLabel: event.projectLabel, repoId: event.repoId, @@ -97,16 +103,19 @@ export function createUsageEventAggregation< reasoningOutputTokens: event.reasoningOutputTokens, totalTokens: event.totalTokens, ...eventMetric - }) + } + target.push(entry) + index.set(event.projectKey, entry) } function foldModel( target: UsageModelBreakdown[], event: TEvent, - eventMetric: TMetric + eventMetric: TMetric, + index: UsageSessionBreakdownIndex['models'] ): void { const key = event.model ?? 'unknown' - const existing = target.find((entry) => entry.modelKey === key) ?? null + const existing = index.get(key) if (existing) { existing.eventCount++ existing.inputTokens += event.inputTokens @@ -118,7 +127,7 @@ export function createUsageEventAggregation< return } - target.push({ + const entry: UsageModelBreakdown = { modelKey: key, modelLabel: event.model ?? 'Unknown model', eventCount: 1, @@ -128,19 +137,19 @@ export function createUsageEventAggregation< reasoningOutputTokens: event.reasoningOutputTokens, totalTokens: event.totalTokens, ...eventMetric - }) + } + target.push(entry) + index.set(key, entry) } function foldLocationModel( target: UsageLocationModelBreakdown[], event: TEvent, - eventMetric: TMetric + eventMetric: TMetric, + index: UsageSessionBreakdownIndex['locationModels'] ): void { const modelKey = event.model ?? 'unknown' - const existing = - target.find( - (entry) => entry.locationKey === event.projectKey && entry.modelKey === modelKey - ) ?? null + const existing = index.get(usageLocationModelKey(event.projectKey, modelKey)) if (existing) { existing.eventCount++ existing.inputTokens += event.inputTokens @@ -152,7 +161,7 @@ export function createUsageEventAggregation< return } - target.push({ + const entry: UsageLocationModelBreakdown = { locationKey: event.projectKey, modelKey, modelLabel: event.model ?? 'Unknown model', @@ -165,7 +174,9 @@ export function createUsageEventAggregation< reasoningOutputTokens: event.reasoningOutputTokens, totalTokens: event.totalTokens, ...eventMetric - }) + } + target.push(entry) + index.set(usageLocationModelKey(event.projectKey, modelKey), entry) } function finalizeSessions( @@ -209,6 +220,7 @@ export function createUsageEventAggregation< } { const sessionsById = new Map>() const dailyByKey = new Map>() + const breakdownsBySession = new Map>() for (const event of events) { const eventMetric = metric.fromEvent(event) @@ -229,9 +241,19 @@ export function createUsageEventAggregation< session.totalReasoningOutputTokens += event.reasoningOutputTokens session.totalTokens += event.totalTokens metric.fold(session, eventMetric) - foldLocation(session.locationBreakdown, event, eventMetric) - foldModel(session.modelBreakdown, event, eventMetric) - foldLocationModel(session.locationModelBreakdown, event, eventMetric) + let breakdowns = breakdownsBySession.get(event.sessionId) + if (!breakdowns) { + breakdowns = indexUsageSessionBreakdowns(session) + breakdownsBySession.set(event.sessionId, breakdowns) + } + foldLocation(session.locationBreakdown, event, eventMetric, breakdowns.locations) + foldModel(session.modelBreakdown, event, eventMetric, breakdowns.models) + foldLocationModel( + session.locationModelBreakdown, + event, + eventMetric, + breakdowns.locationModels + ) const dailyKey = usageDailyAggregateKey(event) const daily = dailyByKey.get(dailyKey) ?? createEmptyDailyAggregate(event) diff --git a/src/main/usage/usage-rollup-merge.ts b/src/main/usage/usage-rollup-merge.ts index f63408f01a1..662a77c07a5 100644 --- a/src/main/usage/usage-rollup-merge.ts +++ b/src/main/usage/usage-rollup-merge.ts @@ -1,4 +1,9 @@ /** Combines rollups produced by separate sources (rollout files, sibling databases) of one provider. */ +import { + indexUsageSessionBreakdowns, + usageLocationModelKey, + type UsageSessionBreakdownIndex +} from './usage-session-breakdown-index' import { usageDailyAggregateKey, type UsageDailyAggregate, @@ -16,6 +21,7 @@ export function mergeUsageSessions( sessions: UsageSession[], { fold, cloneSessionForMerge }: UsageRollupMergeOptions ): void { + const breakdownsBySession = new Map>() for (const session of sessions) { const existing = target.get(session.sessionId) if (!existing) { @@ -39,10 +45,13 @@ export function mergeUsageSessions( existing.totalTokens += session.totalTokens fold(existing, session) + let breakdowns = breakdownsBySession.get(session.sessionId) + if (!breakdowns) { + breakdowns = indexUsageSessionBreakdowns(existing) + breakdownsBySession.set(session.sessionId, breakdowns) + } for (const location of session.locationBreakdown) { - const existingLocation = - existing.locationBreakdown.find((entry) => entry.locationKey === location.locationKey) ?? - null + const existingLocation = breakdowns.locations.get(location.locationKey) if (existingLocation) { existingLocation.eventCount += location.eventCount existingLocation.inputTokens += location.inputTokens @@ -52,13 +61,14 @@ export function mergeUsageSessions( existingLocation.totalTokens += location.totalTokens fold(existingLocation, location) } else { - existing.locationBreakdown.push({ ...location }) + const copy = { ...location } + existing.locationBreakdown.push(copy) + breakdowns.locations.set(location.locationKey, copy) } } for (const model of session.modelBreakdown) { - const existingModel = - existing.modelBreakdown.find((entry) => entry.modelKey === model.modelKey) ?? null + const existingModel = breakdowns.models.get(model.modelKey) if (existingModel) { existingModel.eventCount += model.eventCount existingModel.inputTokens += model.inputTokens @@ -68,17 +78,16 @@ export function mergeUsageSessions( existingModel.totalTokens += model.totalTokens fold(existingModel, model) } else { - existing.modelBreakdown.push({ ...model }) + const copy = { ...model } + existing.modelBreakdown.push(copy) + breakdowns.models.set(model.modelKey, copy) } } for (const locationModel of session.locationModelBreakdown) { - const existingLocationModel = - existing.locationModelBreakdown.find( - (entry) => - entry.locationKey === locationModel.locationKey && - entry.modelKey === locationModel.modelKey - ) ?? null + const existingLocationModel = breakdowns.locationModels.get( + usageLocationModelKey(locationModel.locationKey, locationModel.modelKey) + ) if (existingLocationModel) { existingLocationModel.eventCount += locationModel.eventCount existingLocationModel.inputTokens += locationModel.inputTokens @@ -88,7 +97,12 @@ export function mergeUsageSessions( existingLocationModel.totalTokens += locationModel.totalTokens fold(existingLocationModel, locationModel) } else { - existing.locationModelBreakdown.push({ ...locationModel }) + const copy = { ...locationModel } + existing.locationModelBreakdown.push(copy) + breakdowns.locationModels.set( + usageLocationModelKey(locationModel.locationKey, locationModel.modelKey), + copy + ) } } } diff --git a/src/main/usage/usage-session-breakdown-index.ts b/src/main/usage/usage-session-breakdown-index.ts new file mode 100644 index 00000000000..7cc88be20e5 --- /dev/null +++ b/src/main/usage/usage-session-breakdown-index.ts @@ -0,0 +1,37 @@ +import type { + UsageSession, + UsageLocationBreakdown, + UsageModelBreakdown, + UsageLocationModelBreakdown +} from './usage-rollup-records' + +export function usageLocationModelKey(locationKey: string, modelKey: string): string { + return JSON.stringify([locationKey, modelKey]) +} + +export function indexUsageSessionBreakdowns(session: UsageSession) { + const locations = new Map>() + const models = new Map>() + const locationModels = new Map>() + for (const entry of session.locationBreakdown) { + if (!locations.has(entry.locationKey)) { + locations.set(entry.locationKey, entry) + } + } + for (const entry of session.modelBreakdown) { + if (!models.has(entry.modelKey)) { + models.set(entry.modelKey, entry) + } + } + for (const entry of session.locationModelBreakdown) { + const key = usageLocationModelKey(entry.locationKey, entry.modelKey) + if (!locationModels.has(key)) { + locationModels.set(key, entry) + } + } + return { locations, models, locationModels } +} + +export type UsageSessionBreakdownIndex = ReturnType< + typeof indexUsageSessionBreakdowns +> diff --git a/src/main/warp-themes/directory-entry-order.ts b/src/main/warp-themes/directory-entry-order.ts new file mode 100644 index 00000000000..1b4cbf81bcb --- /dev/null +++ b/src/main/warp-themes/directory-entry-order.ts @@ -0,0 +1,11 @@ +// Warp theme discovery walks user home and app-data trees, so callers filter to +// the entries they want *before* sorting: same resulting order (the comparator is +// a total order over a stable sort), without collating the hundreds of unrelated +// names a home directory holds. +export function sortDirectoryEntriesByName(entries: T[]): T[] { + if (entries.length < 2) { + return entries + } + const compare = new Intl.Collator(undefined, { sensitivity: 'base' }).compare + return entries.sort((left, right) => compare(left.name, right.name)) +} diff --git a/src/main/warp-themes/discovery.test.ts b/src/main/warp-themes/discovery.test.ts index 977a5b30de7..0a6851019c4 100644 --- a/src/main/warp-themes/discovery.test.ts +++ b/src/main/warp-themes/discovery.test.ts @@ -43,6 +43,78 @@ describe('getWarpThemeDirectories', () => { readdirSyncMock.mockReturnValue([]) }) + it('sorts dynamic directories with one collator and preserves locale ties', () => { + platformMock.mockReturnValue('darwin') + const names = ['éclair', 'Eclair', 'item2', 'item10', 'Ångström', 'zebra', 'İstanbul'].map( + (name) => `.warp-${name}` + ) + readdirSyncMock.mockReturnValue(names.map(directoryEntry)) + const expected = [...names].sort((a, b) => + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Preserve the old comparator as the parity oracle. + a.localeCompare(b, undefined, { sensitivity: 'base' }) + ) + const NativeCollator = Intl.Collator + const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + return new NativeCollator(locales, options) + }) + const localeCompare = vi.spyOn(String.prototype, 'localeCompare') + try { + expect(getWarpThemeDirectories().slice(6)).toEqual( + expected.map((name) => `/Users/alice/${name}/themes`) + ) + expect(construct).toHaveBeenCalledExactlyOnceWith(undefined, { sensitivity: 'base' }) + expect(localeCompare).not.toHaveBeenCalled() + } finally { + construct.mockRestore() + localeCompare.mockRestore() + } + }) + + it('collates only the Warp entries in a crowded home directory', () => { + platformMock.mockReturnValue('darwin') + const noise = Array.from({ length: 500 }, (_, index) => directoryEntry(`project-${index}`)) + const warpNames = ['.warp-zebra', '.warp-Ångström', '.warp-éclair'] + readdirSyncMock.mockReturnValue([...noise, ...warpNames.map(directoryEntry)]) + const expected = [...warpNames].sort((a, b) => + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Preserve the old comparator as the parity oracle. + a.localeCompare(b, undefined, { sensitivity: 'base' }) + ) + const NativeCollator = Intl.Collator + const compares: string[][] = [] + vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + const collator = new NativeCollator(locales, options) + return { + ...collator, + compare: (left: string, right: string) => { + compares.push([left, right]) + return collator.compare(left, right) + } + } + }) + try { + expect(getWarpThemeDirectories().slice(6)).toEqual( + expected.map((name) => `/Users/alice/${name}/themes`) + ) + // Filtering first keeps the crowd out of ICU: only Warp names are collated. + expect(compares.flat().every((name) => name.startsWith('.warp'))).toBe(true) + expect(compares.length).toBeLessThan(10) + } finally { + vi.restoreAllMocks() + } + }) + + it('skips a directory scan that finds no dynamic Warp entries', () => { + platformMock.mockReturnValue('darwin') + readdirSyncMock.mockReturnValue([directoryEntry('Documents'), fileEntry('.warprc')]) + const construct = vi.spyOn(Intl, 'Collator') + try { + expect(getWarpThemeDirectories()).toHaveLength(6) + expect(construct).not.toHaveBeenCalled() + } finally { + vi.restoreAllMocks() + } + }) + it('returns macOS Warp channel theme directories in stable-first order', () => { platformMock.mockReturnValue('darwin') expect(getWarpThemeDirectories()).toEqual([ diff --git a/src/main/warp-themes/discovery.ts b/src/main/warp-themes/discovery.ts index fd29419c12c..68f8a70c654 100644 --- a/src/main/warp-themes/discovery.ts +++ b/src/main/warp-themes/discovery.ts @@ -2,6 +2,7 @@ import { readdirSync } from 'node:fs' import type { Dirent } from 'node:fs' import { homedir, platform } from 'node:os' import path from 'node:path' +import { sortDirectoryEntriesByName } from './directory-entry-order' const WARP_CHANNELS = [ { macName: '.warp', linuxName: 'warp-terminal', windowsName: 'Warp' }, @@ -16,10 +17,13 @@ const WARP_CHANNELS = [ } ] -function readDirectoryEntries(directoryPath: string): Dirent[] { +function readDirectoryEntries( + directoryPath: string, + include: (entry: Dirent) => boolean +): Dirent[] { try { - return readdirSync(directoryPath, { withFileTypes: true }).sort((left, right) => - left.name.localeCompare(right.name, undefined, { sensitivity: 'base' }) + return sortDirectoryEntriesByName( + readdirSync(directoryPath, { withFileTypes: true }).filter(include) ) } catch { return [] @@ -57,9 +61,10 @@ function getMacWarpThemeDirectories(home: string): string[] { return warpThemeDirectoriesFromDataHomes( [ ...WARP_CHANNELS.map((channel) => pathImpl.join(home, channel.macName)), - ...readDirectoryEntries(home) - .filter((entry) => entry.isDirectory() && entry.name.startsWith('.warp')) - .map((entry) => pathImpl.join(home, entry.name)) + ...readDirectoryEntries( + home, + (entry) => entry.isDirectory() && entry.name.startsWith('.warp') + ).map((entry) => pathImpl.join(home, entry.name)) ], pathImpl ) @@ -77,13 +82,11 @@ function getLinuxWarpThemeDirectories(home: string): string[] { return warpThemeDirectoriesFromDataHomes( [ ...WARP_CHANNELS.map((channel) => pathImpl.join(dataHome, channel.linuxName)), - ...readDirectoryEntries(dataHome) - .filter( - (entry) => - entry.isDirectory() && - (entry.name === 'warp-terminal' || entry.name.startsWith('warp-')) - ) - .map((entry) => pathImpl.join(dataHome, entry.name)) + ...readDirectoryEntries( + dataHome, + (entry) => + entry.isDirectory() && (entry.name === 'warp-terminal' || entry.name.startsWith('warp-')) + ).map((entry) => pathImpl.join(dataHome, entry.name)) ], pathImpl ) @@ -102,10 +105,7 @@ function getWindowsWarpThemeDirectories(home: string): string[] { path.win32 ) } - for (const entry of readDirectoryEntries(warpAppData)) { - if (!entry.isDirectory()) { - continue - } + for (const entry of readDirectoryEntries(warpAppData, (entry) => entry.isDirectory())) { addDedupeDirectory( directories, seenDirectories, diff --git a/src/main/warp-themes/manual-warp-theme-files.test.ts b/src/main/warp-themes/manual-warp-theme-files.test.ts new file mode 100644 index 00000000000..975e40414dc --- /dev/null +++ b/src/main/warp-themes/manual-warp-theme-files.test.ts @@ -0,0 +1,47 @@ +import path from 'node:path' +import { describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ BrowserWindow: {}, dialog: {} })) + +import { createManualWarpThemeFileCandidates } from './manual-warp-theme-files' + +describe('manual theme ordering', () => { + it('reuses one collator for labels and path ties without changing the selected order', () => { + const names = ['éclair', 'Eclair', 'item2', 'item10', 'Ångström', 'zebra', 'İstanbul'] + const paths = Array.from({ length: 200 }, (_, index) => + path.join('themes', names[index % names.length]!, `${names[(index * 3) % names.length]}.yaml`) + ) + const expected = [...paths].sort( + (a, b) => + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Preserve the old comparator as the parity oracle. + path.basename(a).localeCompare(path.basename(b), undefined, { sensitivity: 'base' }) || + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Preserve the old comparator as the parity oracle. + a.localeCompare(b, undefined, { sensitivity: 'base' }) + ) + const NativeCollator = Intl.Collator + const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + return new NativeCollator(locales, options) + }) + const localeCompare = vi.spyOn(String.prototype, 'localeCompare') + try { + expect(createManualWarpThemeFileCandidates(paths).map((file) => file.path)).toEqual(expected) + expect(construct).toHaveBeenCalledExactlyOnceWith(undefined, { sensitivity: 'base' }) + expect(localeCompare).not.toHaveBeenCalled() + } finally { + vi.restoreAllMocks() + } + }) + + it('returns a single dialog selection without collating', () => { + const construct = vi.spyOn(Intl, 'Collator') + try { + expect(createManualWarpThemeFileCandidates([]).map((file) => file.path)).toEqual([]) + expect(createManualWarpThemeFileCandidates(['a/one.yaml']).map((file) => file.path)).toEqual([ + 'a/one.yaml' + ]) + expect(construct).not.toHaveBeenCalled() + } finally { + vi.restoreAllMocks() + } + }) +}) diff --git a/src/main/warp-themes/manual-warp-theme-files.ts b/src/main/warp-themes/manual-warp-theme-files.ts index 496ad9e3919..b50ff8ee48b 100644 --- a/src/main/warp-themes/manual-warp-theme-files.ts +++ b/src/main/warp-themes/manual-warp-theme-files.ts @@ -2,29 +2,28 @@ import { createHash } from 'node:crypto' import path from 'node:path' import { BrowserWindow, dialog, type OpenDialogOptions, type WebContents } from 'electron' import type { WarpThemeImportSkippedFile } from '../../shared/terminal-custom-themes' -import { - compareThemeFileLabels, - isYamlFile, - MAX_THEME_FILES, - type ThemeFileCandidate -} from './theme-file-scanner' +import { isYamlFile, MAX_THEME_FILES, type ThemeFileCandidate } from './theme-file-scanner' export function createManualWarpThemeFileCandidates(filePaths: string[]): ThemeFileCandidate[] { - return filePaths - .map((filePath) => ({ - path: filePath, - label: path.basename(filePath), - contentHashDiscriminator: true - })) - .sort((left, right) => { - const labelComparison = compareThemeFileLabels(left, right) - if (labelComparison !== 0) { - return labelComparison - } - // Why: manual dialogs can return selections in click order. Sort only in - // main so duplicate basenames get deterministic IDs without persisting paths. - return left.path.localeCompare(right.path, undefined, { sensitivity: 'base' }) - }) + const candidates = filePaths.map((filePath) => ({ + path: filePath, + label: path.basename(filePath), + contentHashDiscriminator: true + })) + // Picking a single file is the common dialog outcome and needs no collation. + if (candidates.length < 2) { + return candidates + } + const compareLabels = new Intl.Collator(undefined, { sensitivity: 'base' }).compare + return candidates.sort((left, right) => { + const labelComparison = compareLabels(left.label, right.label) + if (labelComparison !== 0) { + return labelComparison + } + // Why: manual dialogs can return selections in click order. Sort only in + // main so duplicate basenames get deterministic IDs without persisting paths. + return compareLabels(left.path, right.path) + }) } export function manualWarpThemeContentDiscriminator(label: string, content: string): string { diff --git a/src/main/warp-themes/theme-file-scanner.test.ts b/src/main/warp-themes/theme-file-scanner.test.ts new file mode 100644 index 00000000000..bbfd5639e65 --- /dev/null +++ b/src/main/warp-themes/theme-file-scanner.test.ts @@ -0,0 +1,36 @@ +import { mkdir, mkdtemp, readdir, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import path from 'node:path' +import { expect, it, vi } from 'vitest' +import { scanWarpThemeDirectory } from './theme-file-scanner' + +it('reuses one collator per directory while preserving capped scan order', async () => { + const directory = await mkdtemp(path.join(tmpdir(), 'orca-theme-order-')) + const names = ['éclair', 'item2', 'item10', 'Ångström', 'zebra', 'İstanbul'] + try { + await Promise.all(names.map((name) => writeFile(path.join(directory, `${name}.yaml`), ''))) + await mkdir(path.join(directory, 'nested')) + await writeFile(path.join(directory, 'nested', 'theme.yaml'), '') + const expected = (await readdir(directory)) + // oxlint-disable-next-line sort-comparator-performance/no-repeated-collator -- Preserve the old comparator as the parity oracle. + .sort((a, b) => a.localeCompare(b, undefined, { sensitivity: 'base' })) + .map((name) => (name === 'nested' ? path.join(name, 'theme.yaml') : name)) + const NativeCollator = Intl.Collator + const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { + return new NativeCollator(locales, options) + }) + const localeCompare = vi.spyOn(String.prototype, 'localeCompare') + try { + const result = await scanWarpThemeDirectory(directory, undefined, { themeFileLimit: 6 }) + expect(result.files.map((file) => file.label)).toEqual(expected.slice(0, 6)) + expect(result.themeFileLimitHit).toBe(true) + // One directory needs collation; the single-entry nested folder needs none. + expect(construct).toHaveBeenCalledExactlyOnceWith(undefined, { sensitivity: 'base' }) + expect(localeCompare).not.toHaveBeenCalled() + } finally { + vi.restoreAllMocks() + } + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) diff --git a/src/main/warp-themes/theme-file-scanner.ts b/src/main/warp-themes/theme-file-scanner.ts index ba20abe0cc2..2602acb71da 100644 --- a/src/main/warp-themes/theme-file-scanner.ts +++ b/src/main/warp-themes/theme-file-scanner.ts @@ -2,6 +2,7 @@ import { opendir } from 'node:fs/promises' import type { Dirent } from 'node:fs' import path from 'node:path' import type { WarpThemeImportSkippedFile } from '../../shared/terminal-custom-themes' +import { sortDirectoryEntriesByName } from './directory-entry-order' export const MAX_THEME_FILES = 200 const MAX_THEME_DIRECTORY_DEPTH = 3 @@ -44,17 +45,6 @@ export function isYamlFile(filePath: string): boolean { return YAML_EXTENSIONS.has(path.extname(filePath).toLowerCase()) } -export function compareThemeFileLabels( - left: ThemeFileCandidate, - right: ThemeFileCandidate -): number { - return left.label.localeCompare(right.label, undefined, { sensitivity: 'base' }) -} - -function compareDirentNames(left: Dirent, right: Dirent): number { - return left.name.localeCompare(right.name, undefined, { sensitivity: 'base' }) -} - function isYamlFileEntry(entry: Dirent): boolean { return (entry.isFile() || entry.isSymbolicLink()) && isYamlFile(entry.name) } @@ -141,10 +131,10 @@ async function collectYamlFilesFromDirectory( return } - const sortedEntries = entries.sort(compareDirentNames) if (previewBudgetExpiredWhileReading) { return } + const sortedEntries = sortDirectoryEntriesByName(entries) if (entryLimitHit && !budget.entryLimitReported) { skippedFiles.push({ label: relativeDirectory || sourceLabel, diff --git a/src/main/window/clipboard-ipc-handlers.ts b/src/main/window/clipboard-ipc-handlers.ts index 9528958c25f..9b2bc509e38 100644 --- a/src/main/window/clipboard-ipc-handlers.ts +++ b/src/main/window/clipboard-ipc-handlers.ts @@ -55,8 +55,16 @@ async function saveClipboardImageBufferForTarget( ): Promise { assertClipboardImageByteLengthWithinLimit(buffer.byteLength) const runtimeEnvironmentId = args?.runtimeEnvironmentId?.trim() - if (runtimeEnvironmentId && !args?.connectionId) { - return saveClipboardImageBufferInRuntime(app.getPath('userData'), runtimeEnvironmentId, buffer) + // Why (#17679): with a runtime owner, a connectionId names one of the RUNTIME's SSH + // connections (nested Remote Server -> SSH), not one this process dialed. Looking it up + // in the local provider registry can only miss, so the runtime must perform the save. + if (runtimeEnvironmentId) { + return saveClipboardImageBufferInRuntime( + app.getPath('userData'), + runtimeEnvironmentId, + buffer, + args?.connectionId ?? null + ) } return saveClipboardImageBufferAsTempFile(buffer, args) } diff --git a/src/main/window/clipboard-runtime-image-upload.ts b/src/main/window/clipboard-runtime-image-upload.ts index be66997bd7f..d4d4077466b 100644 --- a/src/main/window/clipboard-runtime-image-upload.ts +++ b/src/main/window/clipboard-runtime-image-upload.ts @@ -27,13 +27,14 @@ async function callRuntimeClipboardMethod( async function saveClipboardImageBase64InRuntime( userDataPath: string, runtimeEnvironmentId: string, - contentBase64: string + contentBase64: string, + connectionId: string | null ): Promise { const startResponse = await callRuntimeEnvironment( userDataPath, runtimeEnvironmentId, 'clipboard.startImageUpload', - { expectedBase64Length: contentBase64.length, connectionId: null }, + { expectedBase64Length: contentBase64.length, connectionId }, CLIPBOARD_IMAGE_SAVE_TIMEOUT_MS ) if (!startResponse.ok) { @@ -45,7 +46,7 @@ async function saveClipboardImageBase64InRuntime( userDataPath, runtimeEnvironmentId, 'clipboard.saveImageAsTempFile', - { contentBase64, connectionId: null } + { contentBase64, connectionId } ) } throw new Error(startResponse.error.message) @@ -104,15 +105,19 @@ async function saveClipboardImageBase64InRuntime( } } +// connectionId: the runtime-side SSH target that holds the workspace, or null for +// the runtime host itself. The runtime resolves it in its own provider registry. export function saveClipboardImageBufferInRuntime( userDataPath: string, runtimeEnvironmentId: string, - buffer: Buffer + buffer: Buffer, + connectionId: string | null = null ): Promise { assertClipboardImageByteLengthWithinLimit(buffer.byteLength) return saveClipboardImageBase64InRuntime( userDataPath, runtimeEnvironmentId, - buffer.toString('base64') + buffer.toString('base64'), + connectionId ) } diff --git a/src/main/window/clipboard-runtime-owned-ssh-paste.test.ts b/src/main/window/clipboard-runtime-owned-ssh-paste.test.ts new file mode 100644 index 00000000000..6e09eb5fd3e --- /dev/null +++ b/src/main/window/clipboard-runtime-owned-ssh-paste.test.ts @@ -0,0 +1,223 @@ +// Nested Remote Orca Server -> SSH image paste (#17679). The REAL ssh-filesystem-dispatch registry +// is used on purpose: the runtime's SSH target is never registered in the client process, so any +// route that consults the local registry fails exactly the way the report did. +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { installFakeAppEnvironment } from '../../../config/scripts/vitest-host-ports-setup' + +const { handleMock, callRuntimeEnvironmentMock, fsWriteFileMock } = vi.hoisted(() => ({ + handleMock: vi.fn(), + callRuntimeEnvironmentMock: vi.fn(), + fsWriteFileMock: vi.fn() +})) + +const PNG = Buffer.from([0, 1, 2, 3]) + +vi.mock('electron', () => ({ + app: { getPath: vi.fn(() => '/tmp') }, + clipboard: { + readImage: () => ({ + getSize: () => ({ height: 1, width: 1 }), + isEmpty: () => false, + toPNG: () => PNG + }), + readText: vi.fn(), + readBuffer: vi.fn(), + writeText: vi.fn(), + writeImage: vi.fn(), + writeBuffer: vi.fn() + }, + ipcMain: { removeHandler: vi.fn(), handle: handleMock }, + nativeImage: { createFromBuffer: vi.fn() } +})) +vi.mock('node:fs/promises', () => ({ + access: vi.fn(), + lstat: vi.fn(), + mkdir: vi.fn(), + opendir: vi.fn().mockRejectedValue(Object.assign(new Error('ENOENT'), { code: 'ENOENT' })), + rm: vi.fn(), + open: vi.fn(), + stat: vi.fn(), + realpath: vi.fn(), + writeFile: fsWriteFileMock, + default: { writeFile: fsWriteFileMock } +})) +vi.mock('../ipc/filesystem-auth', () => ({ + PATH_ACCESS_DENIED_MESSAGE: 'denied', + resolveAuthorizedPath: vi.fn(), + authorizeExternalPath: vi.fn() +})) +vi.mock('../ipc/runtime-environment-transport-routing', () => ({ + callRuntimeEnvironment: callRuntimeEnvironmentMock +})) +vi.mock('./dashboard-popout-window', () => ({ isDashboardPopoutRenderer: () => false })) + +import { registerClipboardHandlers } from './clipboard-ipc-handlers' +import { + getSshFilesystemProvider, + registerSshFilesystemProvider, + SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE, + unregisterSshFilesystemProvider +} from '../providers/ssh-filesystem-dispatch' + +type SaveImageHandler = (event: unknown, args?: unknown) => Promise + +const RUNTIME_ID = 'ubuntu-server' +const RUNTIME_SSH_TARGET = 'jetson' +const CLIENT_SSH_TARGET = 'client-dialed' + +const rendererEvent = { + sender: { + id: 1, + getType: () => 'window', + getURL: () => 'file:///orca/index.html', + isDestroyed: () => false + } +} + +function saveImageHandler(): SaveImageHandler { + handleMock.mockClear() + registerClipboardHandlers({} as never) + const call = handleMock.mock.calls.find((c) => c[0] === 'clipboard:saveImageAsTempFile') + if (!call) { + throw new Error('clipboard:saveImageAsTempFile not registered') + } + return call[1] as SaveImageHandler +} + +function mockRuntimeUpload( + overrides: Record = {} +): void { + callRuntimeEnvironmentMock.mockImplementation(async (_userData, _env, method) => { + if (method in overrides) { + return { ...overrides[method], _meta: { runtimeId: 'r' } } + } + switch (method) { + case 'clipboard.startImageUpload': + return { ok: true, result: { uploadId: 'upload-1' }, _meta: { runtimeId: 'r' } } + case 'clipboard.appendImageUploadChunk': + return { ok: true, result: { receivedBase64Length: 8 }, _meta: { runtimeId: 'r' } } + case 'clipboard.commitImageUpload': + return { ok: true, result: '/tmp/on-runtime-target.png', _meta: { runtimeId: 'r' } } + case 'clipboard.abortImageUpload': + return { ok: true, result: { aborted: true }, _meta: { runtimeId: 'r' } } + default: + throw new Error(`unexpected runtime method ${method}`) + } + }) +} + +function runtimeCall(method: string): unknown[] | undefined { + return callRuntimeEnvironmentMock.mock.calls.find((c) => c[2] === method) +} + +describe('clipboard image paste for a runtime-owned SSH workspace', () => { + beforeEach(() => { + installFakeAppEnvironment({ getPath: () => '/tmp' }) + callRuntimeEnvironmentMock.mockReset() + fsWriteFileMock.mockReset() + unregisterSshFilesystemProvider(RUNTIME_SSH_TARGET) + unregisterSshFilesystemProvider(CLIENT_SSH_TARGET) + }) + + it('sends the paste to the runtime and names the runtime SSH target, never this registry', async () => { + mockRuntimeUpload() + expect(getSshFilesystemProvider(RUNTIME_SSH_TARGET)).toBeUndefined() + // Built the way the renderer builds it: both owner ids, verbatim. + const nestedArgs = { connectionId: RUNTIME_SSH_TARGET, runtimeEnvironmentId: RUNTIME_ID } + + await expect(saveImageHandler()(rendererEvent, nestedArgs)).resolves.toBe( + '/tmp/on-runtime-target.png' + ) + + const start = runtimeCall('clipboard.startImageUpload') + expect(start?.[1]).toBe(nestedArgs.runtimeEnvironmentId) + expect(start?.[3]).toEqual({ + expectedBase64Length: PNG.toString('base64').length, + connectionId: nestedArgs.connectionId + }) + expect(fsWriteFileMock).not.toHaveBeenCalled() + }) + + it('names the runtime SSH target on the single-frame fallback for older runtimes', async () => { + mockRuntimeUpload({ + 'clipboard.startImageUpload': { + ok: false, + error: { code: 'method_not_found', message: 'no such method' } + }, + 'clipboard.saveImageAsTempFile': { ok: true, result: '/tmp/on-runtime-target.png' } + }) + const nestedArgs = { connectionId: RUNTIME_SSH_TARGET, runtimeEnvironmentId: RUNTIME_ID } + + await expect(saveImageHandler()(rendererEvent, nestedArgs)).resolves.toBe( + '/tmp/on-runtime-target.png' + ) + + expect(runtimeCall('clipboard.saveImageAsTempFile')?.[3]).toEqual({ + contentBase64: PNG.toString('base64'), + connectionId: nestedArgs.connectionId + }) + }) + + it('surfaces the runtime verdict instead of the local Reconnect advice when the runtime fails', async () => { + mockRuntimeUpload({ + 'clipboard.startImageUpload': { + ok: false, + error: { code: 'runtime_error', message: 'Unknown environment' } + } + }) + + await expect( + saveImageHandler()(rendererEvent, { + connectionId: RUNTIME_SSH_TARGET, + runtimeEnvironmentId: 'missing-env' + }) + ).rejects.toThrow('Unknown environment') + expect(fsWriteFileMock).not.toHaveBeenCalled() + }) + + it('keeps a runtime-host paste (no SSH target) addressed to the runtime itself', async () => { + mockRuntimeUpload() + + await expect( + saveImageHandler()(rendererEvent, { runtimeEnvironmentId: RUNTIME_ID }) + ).resolves.toBe('/tmp/on-runtime-target.png') + + expect(runtimeCall('clipboard.startImageUpload')?.[3]).toEqual({ + expectedBase64Length: PNG.toString('base64').length, + connectionId: null + }) + }) + + it('keeps a client-dialed SSH paste on this process registry without touching the runtime', async () => { + const writeFileBase64 = vi.fn().mockResolvedValue(undefined) + registerSshFilesystemProvider(CLIENT_SSH_TARGET, { + getTempDir: async () => '/var/tmp', + writeFileBase64 + } as never) + + await expect( + saveImageHandler()(rendererEvent, { connectionId: CLIENT_SSH_TARGET }) + ).resolves.toMatch(/^\/var\/tmp\/orca-paste-.*\.png$/) + + expect(writeFileBase64).toHaveBeenCalledTimes(1) + expect(callRuntimeEnvironmentMock).not.toHaveBeenCalled() + }) + + it('still reports the dropped-connection verdict for a client-dialed SSH target that is gone', async () => { + await expect( + saveImageHandler()(rendererEvent, { connectionId: CLIENT_SSH_TARGET }) + ).rejects.toThrow(SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE) + expect(callRuntimeEnvironmentMock).not.toHaveBeenCalled() + }) + + it('keeps a plain local paste on the local temp dir', async () => { + fsWriteFileMock.mockResolvedValue(undefined) + + await expect(saveImageHandler()(rendererEvent, undefined)).resolves.toMatch( + /orca-paste-.*\.png$/ + ) + + expect(fsWriteFileMock).toHaveBeenCalledTimes(1) + expect(callRuntimeEnvironmentMock).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/window/runtime-window-lifecycle.ts b/src/main/window/runtime-window-lifecycle.ts index 2a6ab95a07f..78c5f2ec426 100644 --- a/src/main/window/runtime-window-lifecycle.ts +++ b/src/main/window/runtime-window-lifecycle.ts @@ -149,8 +149,6 @@ export function registerRuntimeWindowLifecycle( resolution, ...(ptyId ? { ptyId } : {}) }), - setLegacyWorkerTerminalResumeFence: (paneKey, blocked) => - send('agentStatus:legacyWorkerTerminalResumeFence', { paneKey, blocked }), splitTerminal: (tabId, paneRuntimeId, opts) => { send('ui:splitTerminal', { tabId, diff --git a/src/preload/api/agent-status-api.ts b/src/preload/api/agent-status-api.ts index 89677022506..7aa6c21115d 100644 --- a/src/preload/api/agent-status-api.ts +++ b/src/preload/api/agent-status-api.ts @@ -28,10 +28,6 @@ export type AgentStatusApi = { ptyId?: string }) => void ) => () => void - /** Listen for the automatic-resume fence a settled worker's pane gains or loses mid-session. */ - onLegacyWorkerTerminalResumeFence: ( - callback: (data: { paneKey: string; blocked: boolean }) => void - ) => () => void getMigrationUnsupportedSnapshot: () => Promise /** Drop a paneKey from the main-process hook cache and on-disk last-status file. Fire-and-forget. */ drop: (paneKey: string) => void diff --git a/src/preload/api/agent-status-bridge.ts b/src/preload/api/agent-status-bridge.ts index 3c3415cd207..3cc1654aaed 100644 --- a/src/preload/api/agent-status-bridge.ts +++ b/src/preload/api/agent-status-bridge.ts @@ -61,16 +61,6 @@ export const agentStatusApi = { ipcRenderer.on('agentStatus:legacyWorkerTerminalRecovery', listener) return () => ipcRenderer.removeListener('agentStatus:legacyWorkerTerminalRecovery', listener) }, - onLegacyWorkerTerminalResumeFence: ( - callback: (data: { paneKey: string; blocked: boolean }) => void - ): (() => void) => { - const listener = ( - _event: Electron.IpcRendererEvent, - data: { paneKey: string; blocked: boolean } - ) => callback(data) - ipcRenderer.on('agentStatus:legacyWorkerTerminalResumeFence', listener) - return () => ipcRenderer.removeListener('agentStatus:legacyWorkerTerminalResumeFence', listener) - }, getMigrationUnsupportedSnapshot: (): Promise => ipcRenderer.invoke('agentStatus:getMigrationUnsupportedSnapshot'), /** Drop the cached hook status for a paneKey on both sides (memory + on-disk) so a relaunch can't resurrect a dismissed row. */ diff --git a/src/preload/api/pty-api.ts b/src/preload/api/pty-api.ts index d2d547d98cb..f7ce3dfc1af 100644 --- a/src/preload/api/pty-api.ts +++ b/src/preload/api/pty-api.ts @@ -4,7 +4,7 @@ import type { } from '../../shared/agent-session-resume' import type { StartupCommandDelivery } from '../../shared/codex-startup-delivery' import type { ProjectExecutionRuntimeResolution } from '../../shared/project-execution-runtime' -import type { PtyListedSession } from '../../shared/pty-listed-session' +import type { PtyListedSession, PtySessionListScope } from '../../shared/pty-listed-session' import type { PtyMainDeliveryDiagnostics } from '../../shared/pty-delivery-diagnostics' import type { PtyModelRestoreNeededEvent } from '../../shared/pty-model-restore-marker' import type { @@ -123,7 +123,7 @@ export type PtyApi = { confirmForegroundProcess: (id: string) => Promise getCwd: (id: string) => Promise getSize: (id: string) => Promise<{ cols: number; rows: number } | null> - listSessions: () => Promise + listSessions: (scope?: PtySessionListScope) => Promise getAuthoritativeBufferSnapshotCapabilities?: ( ids: string[] ) => Promise<{ id: string; authoritative: boolean | null }[]> diff --git a/src/preload/api/pty-bridge-session-control.ts b/src/preload/api/pty-bridge-session-control.ts index 2854278c20b..d85967c6f23 100644 --- a/src/preload/api/pty-bridge-session-control.ts +++ b/src/preload/api/pty-bridge-session-control.ts @@ -7,7 +7,7 @@ import type { SleepingAgentLaunchConfig } from '../../shared/agent-session-resume' import type { TuiAgent } from '../../shared/tui-agent' -import type { PtyListedSession } from '../../shared/pty-listed-session' +import type { PtyListedSession, PtySessionListScope } from '../../shared/pty-listed-session' import type { PtyRendererDeliveryHealthReply, PtyRendererDeliveryStateReport @@ -150,7 +150,8 @@ export const ptySessionControlApi = { }, kill: (id: string, opts?: { keepHistory?: boolean }): Promise => ipcRenderer.invoke('pty:kill', { id, keepHistory: opts?.keepHistory ?? false }), - listSessions: (): Promise => ipcRenderer.invoke('pty:listSessions'), + listSessions: (scope?: PtySessionListScope): Promise => + ipcRenderer.invoke('pty:listSessions', scope), getAuthoritativeBufferSnapshotCapabilities: ( ids: string[] ): Promise<{ id: string; authoritative: boolean | null }[]> => diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index 3527f11c44e..a65dfe59e9a 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -400,11 +400,6 @@ z-index: 40 !important; } -/* Above the z-40 updater/onboarding chrome, below the floating workspace panel's z-45. */ -.native-chat-pane-shell:has([data-native-chat-working='true']) { - z-index: 44; -} - [data-sonner-toaster] [data-sonner-toast][data-styled='true'] { align-items: flex-start; flex-wrap: wrap; @@ -1514,18 +1509,16 @@ html.native-shell .app-layout { grid-template-rows: 1fr; } -/* Why: must stay a compositor-driven CSS animation — a JS clock writing - per-element transforms blocks the renderer input thread (STA-3328: 41 - spinners ⇒ ~490 style writes/s, keystroke p90 363ms). The component anchors - each animation's startTime once; steps(12) preserves the retired cadence. */ +/* Keep rotation on the compositor: JS style writes caused typing stalls (#12359). */ +/* One day per iteration avoids per-second React animation events; still 12 steps/second. */ @keyframes agent-spinner-rotate { to { - transform: rotate(360deg); + transform: rotate(86400turn); } } .agent-working-spinner { - animation: agent-spinner-rotate 1s steps(12, end) infinite; + animation: agent-spinner-rotate 86400s steps(1036800, end) infinite; } @media (prefers-reduced-motion: reduce) { diff --git a/src/renderer/src/components/AgentWorkingSpinner.test.tsx b/src/renderer/src/components/AgentWorkingSpinner.test.tsx index f887dab132c..c2aca2c359b 100644 --- a/src/renderer/src/components/AgentWorkingSpinner.test.tsx +++ b/src/renderer/src/components/AgentWorkingSpinner.test.tsx @@ -183,15 +183,13 @@ describe('AgentWorkingSpinner', () => { } }) - // Why: the class only spins if main.css defines it — pin the wiring across - // both files so neither side can be renamed or dropped alone (STA-3328 - // regressed typing latency when rotation moved onto the input thread). - it('is backed by a steps(12) keyframe animation in main.css', () => { + it('preserves 12 steps per second without frequent iteration events', () => { const css = readFileSync(join(__dirname, '../assets/main.css'), 'utf8') const rule = css.match(/\.agent-working-spinner\s*\{[^}]*\}/)?.[0] expect(rule).toBeDefined() - expect(rule).toContain('animation: agent-spinner-rotate 1s steps(12, end) infinite') + expect(rule).toContain('animation: agent-spinner-rotate 86400s steps(1036800, end) infinite') + expect(css).toContain('transform: rotate(86400turn)') expect(css).toContain('@keyframes agent-spinner-rotate') const reducedMotionBlock = css.match( diff --git a/src/renderer/src/components/Landing.tsx b/src/renderer/src/components/Landing.tsx index d03614b63dd..629836c58f6 100644 --- a/src/renderer/src/components/Landing.tsx +++ b/src/renderer/src/components/Landing.tsx @@ -280,7 +280,7 @@ export default function Landing(): React.JSX.Element { onClick={() => openModal('add-repo')} > - {translate('auto.components.Landing.f9eaa9e12d', 'Add Project')} + {translate('auto.components.Landing.f9eaa9e12d', 'Add project')}
- {translate( - 'auto.components.activity.ActivityPrototypePage.showSearch', - 'Show search' - )} - - {showSearch ? : null} + {translate( + 'auto.components.activity.ActivityPrototypePage.showUnreadOnly', + 'Show unread only' + )} ) : null} - {onToggleUnread ? ( - - - onToggleUnread()} - onSelect={(event) => event.preventDefault()} - > - - - {translate( - 'auto.components.activity.ActivityPrototypePage.showUnreadOnly', - 'Show unread only' - )} - - {unreadOnly ? : null} - - - - {translate( - 'auto.components.activity.ActivityPrototypePage.unreadOnlyDescription', - 'Filters the activity list to show only threads with unread updates.' - )} - - + {onShowChildAgentsChange ? ( + onShowChildAgentsChange(checked === true)} + onSelect={(event) => event.preventDefault()} + > + {translate( + 'auto.components.activity.ActivityPrototypePage.showChildAgents', + 'Show child agents' + )} + ) : null} + ) : null} - + + {translate('auto.components.activity.ActivityPrototypePage.viewSection', 'View')} + {groupBy && onGroupByChange ? ( - - + {translate( 'auto.components.activity.ActivityPrototypePage.770d458144', 'Group by' )} - + {getActivityGroupByLabel(groupBy)} @@ -207,73 +195,36 @@ export function ActivityThreadOptionsMenu({ value={groupBy} onValueChange={(value) => onGroupByChange(value as ActivityGroupBy)} > - {[ - ['none', 'None', 'auto.components.activity.ActivityPrototypePage.none'], - ['status', 'Status', 'auto.components.activity.ActivityPrototypePage.4a3986b200'], - [ - 'project', - 'Project', - 'auto.components.activity.ActivityPrototypePage.8c3b621ddf' - ], - [ - 'worktree', - 'Worktree', - 'auto.components.activity.ActivityPrototypePage.b29191b3e0' - ], - ['agent', 'Agent', 'auto.components.activity.ActivityPrototypePage.f6396e1f85'] - ].map(([value, label, key]) => ( + {GROUP_BY_OPTIONS.map((value) => ( event.preventDefault()} > - {translate(key, label)} + {getActivityGroupByLabel(value)} ))} ) : null} - - - onCompactModeChange(checked === true)} - onSelect={(event) => event.preventDefault()} - > - - - {translate( - 'auto.components.activity.ActivityPrototypePage.f70e4bec47', - 'Compact mode' - )} - - {compactMode ? : null} - - - - {translate( - 'auto.components.activity.ActivityPrototypePage.compactModeDescription', - 'Shows shorter thread rows with one-line titles and two-line status messages.' - )} - - - {onShowChildAgentsChange ? ( + onCompactModeChange(checked === true)} + onSelect={(event) => event.preventDefault()} + > + {translate('auto.components.activity.ActivityPrototypePage.f70e4bec47', 'Compact mode')} + + {onShowSearchChange ? ( onShowChildAgentsChange(checked === true)} - onSelect={(event) => event.preventDefault()} + checked={showSearch} + onCheckedChange={(checked) => { + const show = checked === true + skipCloseAutoFocusRef.current = show + onShowSearchChange(show) + }} > - - - {translate( - 'auto.components.activity.ActivityPrototypePage.showChildAgents', - 'Show child agents' - )} - - {showChildAgents ? : null} + {translate('auto.components.activity.ActivityPrototypePage.showSearch', 'Show search')} ) : null} {onMarkAllThreadsRead || onClearCompleted ? ( diff --git a/src/renderer/src/components/activity/activity-thread-presentation.ts b/src/renderer/src/components/activity/activity-thread-presentation.ts index f1262082345..1aa27d321dc 100644 --- a/src/renderer/src/components/activity/activity-thread-presentation.ts +++ b/src/renderer/src/components/activity/activity-thread-presentation.ts @@ -59,6 +59,9 @@ export function activityThreadResponseRenderPreview({ } export function agentTitle(event: ActivityEvent): string { + if (event.state === 'working') { + return 'Agent working' + } if (event.state === 'done') { return event.entry.interrupted ? 'Agent interrupted' : 'Agent finished' } @@ -67,6 +70,9 @@ export function agentTitle(event: ActivityEvent): string { export function agentSummary(event: ActivityEvent): string { const prompt = getAgentRowPrimaryText(event.entry) + if (event.state === 'working') { + return prompt || 'The agent is working on the current turn.' + } if (event.state === 'done') { const message = event.entry.lastAssistantMessage?.trim() return message || prompt || 'Completed the current turn.' @@ -76,6 +82,9 @@ export function agentSummary(event: ActivityEvent): string { export function agentMeta(event: ActivityEvent): string { const agent = formatAgentTypeLabel(event.agentType) + if (event.state === 'working') { + return `${agent} ${event.state}` + } if (event.state === 'done') { return event.entry.interrupted ? `${agent} interrupted` : `${agent} completed` } @@ -103,18 +112,28 @@ export function statusPreviewForEntry( return resolveActivityThreadStatusPreview(entry, agentState, previousPreview) } +export type ActivityThreadStatusId = AgentDotState + +/** Single classifier behind grouping, labels, and clear-completed; the only place the + * interrupted predicate is spelled. */ +export function activityThreadStatusId(thread: AgentPaneThread): ActivityThreadStatusId { + const state = thread.currentAgentState ?? thread.latestEvent?.state ?? 'done' + if (!thread.currentAgentState && state === 'done' && thread.latestEvent?.entry.interrupted) { + return 'interrupted' + } + return state +} + +// Interrupted rows deliberately keep the done glyph (#2569). export function threadAgentState(thread: AgentPaneThread): AgentDotState { - return thread.currentAgentState ?? thread.latestEvent?.state ?? 'done' + const id = activityThreadStatusId(thread) + return id === 'interrupted' ? 'done' : id } export function threadAgentStateLabel(thread: AgentPaneThread): string { - const state = threadAgentState(thread) - if (!thread.currentAgentState && state === 'done' && thread.latestEvent?.entry.interrupted) { - return translate('auto.components.activity.ActivityPrototypePage.interrupted', 'Interrupted') - } // Literal keys with literal fallbacks: a dynamic key registers no catalog reference // and forces every state string into the boot bundle. - switch (state) { + switch (activityThreadStatusId(thread)) { case 'working': return translate('auto.components.activity.ActivityPrototypePage.state.working', 'Working') case 'monitoring': diff --git a/src/renderer/src/components/activity/activity-thread-types.ts b/src/renderer/src/components/activity/activity-thread-types.ts index ed73642c8eb..7d32f1a2b39 100644 --- a/src/renderer/src/components/activity/activity-thread-types.ts +++ b/src/renderer/src/components/activity/activity-thread-types.ts @@ -11,20 +11,12 @@ import type { ActivityPortalReadinessStatus } from './activity-portal-readiness- export type { ActivityGroupBy, ThreadReadFilter } from '../../../../shared/ui-chrome-types' -export type ActivityEventState = Extract +export type ActivityEventState = AgentStatusState export type ActivityHookLiveAgentState = Extract< AgentStatusState, 'working' | 'blocked' | 'waiting' > export type ActivityLiveAgentState = ActivityHookLiveAgentState | 'monitoring' -export type ActivityStatusGroupId = - | 'working' - | 'monitoring' - | 'blocked' - | 'waiting' - | 'done' - | 'interrupted' - export type ActivityEvent = { id: string state: ActivityEventState @@ -69,7 +61,6 @@ export type AgentPaneThread = { export type ActivityThreadGroup = { key: string - id?: ActivityStatusGroupId label: string state?: AgentDotState threads: AgentPaneThread[] diff --git a/src/renderer/src/components/activity/useActivityUnreadCount.freshness.test.tsx b/src/renderer/src/components/activity/useActivityUnreadCount.freshness.test.tsx new file mode 100644 index 00000000000..b8161c50057 --- /dev/null +++ b/src/renderer/src/components/activity/useActivityUnreadCount.freshness.test.tsx @@ -0,0 +1,81 @@ +// @vitest-environment happy-dom +import { act, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { AGENT_STATUS_STALE_AFTER_MS } from '../../../../shared/agent-status-types' + +// Real slice, not a hand-rolled store: the hook subscribes to agentStatusEpoch alone, so the +// test must prove the reducer bumps that epoch for every transition the count depends on. +vi.mock('@/store', async () => { + const { createTestStore } = await import('@/store/slices/store-test-helpers') + return { useAppStore: createTestStore() } +}) + +import { useAppStore } from '@/store' +import { flushMicrotasks } from '@/store/slices/agent-status-test-harness' +import { useActivityUnreadCount } from './useActivityUnreadCount' + +const PANE_KEY = 'tab-1:11111111-1111-4111-8111-111111111111' +const START = 2_000 + +beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(START) +}) +afterEach(() => { + useAppStore.getState().removeAgentStatus(PANE_KEY) + vi.useRealTimers() +}) + +function setWorking(updatedAt: number): void { + useAppStore + .getState() + .setAgentStatus( + PANE_KEY, + { state: 'working', prompt: 'Fix tests', agentType: 'claude' }, + undefined, + { updatedAt, evidenceObservedAt: updatedAt } + ) +} + +describe('useActivityUnreadCount freshness invalidation', () => { + it('ignores same-turn heartbeats, decays at the stale boundary, revives on the next heartbeat', async () => { + setWorking(START) + const hook = renderHook(() => useActivityUnreadCount()) + expect(hook.result.current).toBe(1) + const epochAfterStart = useAppStore.getState().agentStatusEpoch + + // Fresh same-turn heartbeat: no epoch bump, count unchanged. + act(() => { + vi.setSystemTime(START + 1_000) + setWorking(START + 1_000) + }) + expect(useAppStore.getState().agentStatusEpoch).toBe(epochAfterStart) + expect(hook.result.current).toBe(1) + + // Freshness scheduler bumps the epoch at the stale boundary; the count decays with no write. + await act(async () => { + await flushMicrotasks() + vi.advanceTimersByTime(AGENT_STATUS_STALE_AFTER_MS + 1) + }) + expect(hook.result.current).toBe(0) + + // A heartbeat on a stale entry is sort-relevant, so the reducer bumps the epoch and revives it. + const revivedAt = Date.now() + act(() => { + setWorking(revivedAt) + }) + expect(hook.result.current).toBe(1) + + // Reading it holds across further heartbeats. + act(() => { + useAppStore.getState().acknowledgeAgents([PANE_KEY]) + }) + expect(hook.result.current).toBe(0) + act(() => { + vi.setSystemTime(revivedAt + 1_000) + setWorking(revivedAt + 1_000) + }) + expect(hook.result.current).toBe(0) + hook.unmount() + }) +}) diff --git a/src/renderer/src/components/activity/useActivityUnreadCount.test.ts b/src/renderer/src/components/activity/useActivityUnreadCount.test.ts index 62c423ce1bd..b2ab75e9564 100644 --- a/src/renderer/src/components/activity/useActivityUnreadCount.test.ts +++ b/src/renderer/src/components/activity/useActivityUnreadCount.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { + AGENT_STATUS_STALE_AFTER_MS, + type AgentStatusEntry +} from '../../../../shared/agent-status-types' import { countActivityUnread } from './useActivityUnreadCount' const PANE = 'tab-1:11111111-1111-4111-8111-111111111111' @@ -109,3 +112,56 @@ describe('countActivityUnread source overlap', () => { expect(countActivityUnread(source)).toBe(1) }) }) + +describe('countActivityUnread working turns', () => { + it('counts fresh working, but not monitoring, historical, or retained working', () => { + const entry = makeEntry({ + state: 'working', + stateHistory: [{ state: 'working', prompt: 'old', startedAt: 1_000 }] + }) + expect(countActivityUnread(makeSource(entry), 2_000)).toBe(1) + // Monitoring emits no unread event in the list (4b2e3dded0), so the badge must not count it. + expect(countActivityUnread(makeSource({ ...entry, workingMode: 'monitoring' }), 2_000)).toBe(0) + expect( + countActivityUnread( + { + ...makeSource(entry), + agentStatusByPaneKey: {}, + retainedAgentsByPaneKey: { + [PANE]: { + entry, + worktreeId: 'wt-1', + tab: {} as never, + agentType: 'claude', + startedAt: 1_000 + } + } + }, + 2_000 + ) + ).toBe(0) + }) + + it('preserves receipts across heartbeats and counts the next turn', () => { + const entry = makeEntry({ state: 'working', updatedAt: 3_000 }) + expect(countActivityUnread(makeSource(entry, 2_000), 3_000)).toBe(0) + expect(countActivityUnread(makeSource({ ...entry, stateStartedAt: 3_000 }, 2_000), 3_000)).toBe( + 1 + ) + expect( + countActivityUnread( + { ...makeSource(entry), activityClearedAtByPaneKey: { [PANE]: 2_000 } }, + 3_000 + ) + ).toBe(0) + }) + + it('drops stale or unconfirmed working and revives only on fresh evidence', () => { + const entry = makeEntry({ state: 'working' }) + expect(countActivityUnread(makeSource(entry), 2_000 + AGENT_STATUS_STALE_AFTER_MS)).toBe(1) + const expiredAt = 2_001 + AGENT_STATUS_STALE_AFTER_MS + expect(countActivityUnread(makeSource(entry), expiredAt)).toBe(0) + expect(countActivityUnread(makeSource({ ...entry, updatedAt: expiredAt }), expiredAt)).toBe(1) + expect(countActivityUnread(makeSource({ ...entry, restoredUnconfirmed: true }), 2_000)).toBe(0) + }) +}) diff --git a/src/renderer/src/components/activity/useActivityUnreadCount.ts b/src/renderer/src/components/activity/useActivityUnreadCount.ts index 3a3074d24e9..e727ef969e2 100644 --- a/src/renderer/src/components/activity/useActivityUnreadCount.ts +++ b/src/renderer/src/components/activity/useActivityUnreadCount.ts @@ -4,7 +4,9 @@ import { useShallow } from 'zustand/react/shallow' import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' import { useAppStore } from '@/store' import type { AppState } from '@/store/types' -import type { AgentStatusEntry, AgentStatusState } from '../../../../shared/agent-status-types' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' + +import { freshActivityLiveAgentState, isHistoricalActivityState } from './activity-event-state' type ActivityUnreadCountSource = Pick< AppState, @@ -17,18 +19,14 @@ type ActivityUnreadCountSource = Pick< activityClearedAtByPaneKey?: Record } -function isUnreadAgentState(state: AgentStatusState): boolean { - return state === 'done' || state === 'blocked' || state === 'waiting' -} - -/** Counts unread done/blocked/waiting events for the Activity page titlebar badge. */ -export function countActivityUnread(source: ActivityUnreadCountSource): number { +/** Counts unread historical activity and fresh current turns. */ +export function countActivityUnread(source: ActivityUnreadCountSource, now = Date.now()): number { let count = 0 const seenPaneKeys = new Set() // Why no worktree.isUnread here: Activity lists only agent threads, so a worktree // unread would light a badge with no row to read and no way to clear it. - const countEntry = (entry: AgentStatusEntry, ackAt: number): void => { + const countEntry = (entry: AgentStatusEntry, ackAt: number, live = false): void => { // Why: "Clear completed" hides events at or before the pane's cutoff from the feed, // so a hidden event must not keep the badge lit; treat the cutoff like an ack floor. const clearedAt = source.activityClearedAtByPaneKey?.[entry.paneKey] ?? 0 @@ -36,13 +34,16 @@ export function countActivityUnread(source: ActivityUnreadCountSource): number { // Why: Activity feed surfaces historical done/blocked/waiting events // from stateHistory, so the titlebar badge must mirror that event count. for (const history of entry.stateHistory) { - if (isUnreadAgentState(history.state) && mutedAt < history.startedAt) { + if (isHistoricalActivityState(history.state) && mutedAt < history.startedAt) { count += 1 } } // Why: a session-boundary done is an idle connect (STA-3386), not an event to read. + // Why 'working' only: a monitoring turn surfaces through the live snapshot, never as an + // unread event, so counting it here would light the badge with no unread row to clear. if ( - isUnreadAgentState(entry.state) && + (isHistoricalActivityState(entry.state) || + (live && freshActivityLiveAgentState(entry, now) === 'working')) && entry.sessionBoundary !== true && mutedAt < entry.stateStartedAt ) { @@ -52,7 +53,7 @@ export function countActivityUnread(source: ActivityUnreadCountSource): number { for (const [paneKey, entry] of Object.entries(source.agentStatusByPaneKey)) { seenPaneKeys.add(paneKey) - countEntry(entry, source.acknowledgedAgentsByPaneKey[paneKey] ?? 0) + countEntry(entry, source.acknowledgedAgentsByPaneKey[paneKey] ?? 0, true) } for (const [paneKey, retained] of Object.entries(source.retainedAgentsByPaneKey)) { // Live status is the primary source; retained is a handoff cache and may briefly overlap it. @@ -75,17 +76,17 @@ export function countActivityUnread(source: ActivityUnreadCountSource): number { export function useActivityUnreadCount(): number { const { - sortEpoch, + agentStatusEpoch, migrationUnsupportedByPtyId, retainedAgentsByPaneKey, acknowledgedAgentsByPaneKey, activityClearedAtByPaneKey } = useAppStore( useShallow((state) => ({ - // Why: live status prompt/tool updates churn agentStatusByPaneKey but - // cannot change unread count unless a sort-relevant state transition - // or removal occurred. sortEpoch is the cheap invalidation signal. - sortEpoch: state.sortEpoch, + // Why not the status map: the receipt is keyed on stateStartedAt, so same-turn heartbeats + // cannot change the count. The live reducer bumps this epoch on state/turn changes and when + // a stale entry revives; the freshness scheduler bumps it at the stale boundary. + agentStatusEpoch: state.agentStatusEpoch, migrationUnsupportedByPtyId: state.migrationUnsupportedByPtyId, retainedAgentsByPaneKey: state.retainedAgentsByPaneKey, acknowledgedAgentsByPaneKey: state.acknowledgedAgentsByPaneKey, @@ -94,7 +95,7 @@ export function useActivityUnreadCount(): number { ) return useMemo(() => { - void sortEpoch + void agentStatusEpoch return countActivityUnread({ agentStatusByPaneKey: useAppStore.getState().agentStatusByPaneKey, migrationUnsupportedByPtyId, @@ -107,6 +108,6 @@ export function useActivityUnreadCount(): number { activityClearedAtByPaneKey, migrationUnsupportedByPtyId, retainedAgentsByPaneKey, - sortEpoch + agentStatusEpoch ]) } diff --git a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPaneOverlayLayer.tsx b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPaneOverlayLayer.tsx index 3aee63c0af7..b15c8420ff0 100644 --- a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPaneOverlayLayer.tsx +++ b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPaneOverlayLayer.tsx @@ -15,6 +15,7 @@ import { useClientHostedBrowserRows } from '@/lib/pane-manager/client-hosted-browser-row-state' import { ClientHostedBrowserHostRowPane } from '../client-hosted-browser-host-row-pane' +import { useAnyBrowserPageMountAdmission } from '../host-guest/browser-page-mount-admission' // Why: Electron destroys its guest on DOM reparent, so BrowserPanes render at worktree level and moving a tab between groups only swaps the overlay's CSS position-anchor. @@ -60,7 +61,8 @@ const BrowserOverlaySlot = memo(function BrowserOverlaySlot({ ? browserTab.pageIds : [browserTab.activePageId ?? browserTab.id] const needsGuestPaint = useBrowserGuestPaintRetention(browserPageIds) - const isPaintable = isActive || needsGuestPaint + const isMountAdmitted = useAnyBrowserPageMountAdmission(browserPageIds) + const isPaintable = isActive || needsGuestPaint || isMountAdmitted // Why: CSS anchor positioning pins the overlay to its owning group's body — a tab move only swaps positionAnchor, no measurement/state. // Orphan branch (no anchorName) stays display:none until the tab is reassigned or destroyed. const style: React.CSSProperties = useMemo( diff --git a/src/renderer/src/components/browser-pane/assemble-chrome/browser-workspace-pane.tsx b/src/renderer/src/components/browser-pane/assemble-chrome/browser-workspace-pane.tsx index b50a6aa1692..e473f442a90 100644 --- a/src/renderer/src/components/browser-pane/assemble-chrome/browser-workspace-pane.tsx +++ b/src/renderer/src/components/browser-pane/assemble-chrome/browser-workspace-pane.tsx @@ -22,6 +22,10 @@ import { WorkspaceDocPagePane } from '../workspace-doc/workspace-doc-page-pane' import { DeferredBrowserContent } from './DeferredBrowserContent' import { isBrowserPagePanePaintable } from '../host-guest/browser-page-paintability' import { SshRoutedBrowserPageGate } from './ssh-routed-browser-page-gate' +import { + isBrowserPageMountAdmitted, + useAnyBrowserPageMountAdmission +} from '../host-guest/browser-page-mount-admission' export default function BrowserPane({ browserTab, @@ -71,6 +75,7 @@ export default function BrowserPane({ () => localBrowserPages.map((page) => page.id), [localBrowserPages] ) + const hasAdmittedPage = useAnyBrowserPageMountAdmission(localBrowserPageIds) const pageDriver = useBrowserDriverForPage(activeBrowserPageId) // Why: a runtime-backed page is streamed, never locally driven, so its driver must read idle. const activeBrowserDriver = runtimeEnvironmentActive ? IDLE_BROWSER_DRIVER : pageDriver @@ -166,7 +171,9 @@ export default function BrowserPane({ key={page.id} retainMounted={isWorktreeActive} mountEligible={isBrowserPagePanePaintable({ - isActive: isActive && page.id === activeBrowserPageId, + isActive: + (isActive && page.id === activeBrowserPageId) || + (hasAdmittedPage && isBrowserPageMountAdmitted(page.id)), isAutomationVisible: automationVisiblePageIds.has(page.id), isMobileDriven: mobileDrivenPageIds.has(page.id), hasRemoteViewer: remotelyViewedPageIds.has(page.id) diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-mount-admission.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-mount-admission.ts new file mode 100644 index 00000000000..3c2b1c8a706 --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-mount-admission.ts @@ -0,0 +1,63 @@ +import { useSyncExternalStore } from 'react' + +// Newly requested pages must start a guest even when opened in the background. Restored pages are +// deliberately absent so worktree restoration can remain lazy. +const admittedPageIds = new Set() +const listeners = new Set<() => void>() +let version = 0 + +export function isBrowserPageMountAdmitted(pageId: string): boolean { + return admittedPageIds.has(pageId) +} + +function emit(): void { + version += 1 + for (const listener of listeners) { + listener() + } +} + +export function admitBrowserPageMount(pageId: string): void { + if (admittedPageIds.has(pageId)) { + return + } + admittedPageIds.add(pageId) + emit() +} + +export function releaseBrowserPageMount(pageId: string): void { + if (!admittedPageIds.delete(pageId)) { + return + } + emit() +} + +export function useBrowserPageMountAdmission(pageId: string): boolean { + useSyncExternalStore( + (listener) => { + listeners.add(listener) + return () => listeners.delete(listener) + }, + () => { + void version + return isBrowserPageMountAdmitted(pageId) + }, + () => false + ) + return isBrowserPageMountAdmitted(pageId) +} + +export function useAnyBrowserPageMountAdmission(pageIds: readonly string[]): boolean { + useSyncExternalStore( + (listener) => { + listeners.add(listener) + return () => listeners.delete(listener) + }, + () => { + void version + return pageIds.some(isBrowserPageMountAdmitted) + }, + () => false + ) + return pageIds.some(isBrowserPageMountAdmitted) +} diff --git a/src/renderer/src/components/dashboard-popout/preview-terminal-options.test.ts b/src/renderer/src/components/dashboard-popout/preview-terminal-options.test.ts index 319ec241293..2ad4147d235 100644 --- a/src/renderer/src/components/dashboard-popout/preview-terminal-options.test.ts +++ b/src/renderer/src/components/dashboard-popout/preview-terminal-options.test.ts @@ -58,6 +58,43 @@ describe('buildPreviewTerminalOptions', () => { scrollback: 1000 } + // #10754: the dashboard preview renders the agent's live buffer, so it has to reproduce the same + // contrast floor the pane used or a Powerline statusline looks different in the popout. + it('mirrors the automatic contrast floor when no override is set', () => { + expect( + buildPreviewTerminalOptions({ + ...base, + terminalInput: null, + theme: { background: '#1e242a' } + }).minimumContrastRatio + ).toBe(3) + expect( + buildPreviewTerminalOptions({ + ...base, + terminalInput: null, + theme: { background: '#ffffff' }, + themeMode: 'light' + }).minimumContrastRatio + ).toBe(4.5) + }) + + it('honors the user contrast override, clamped to xterm range', () => { + expect( + buildPreviewTerminalOptions({ + ...base, + terminalInput: null, + settings: { ...SETTINGS, terminalMinimumContrastRatio: 1 } + }).minimumContrastRatio + ).toBe(1) + expect( + buildPreviewTerminalOptions({ + ...base, + terminalInput: null, + settings: { ...SETTINGS, terminalMinimumContrastRatio: 0 } + }).minimumContrastRatio + ).toBe(1) + }) + it('keeps the kitty advertisement and skips ConPTY options off Windows', () => { const options = buildPreviewTerminalOptions({ ...base, diff --git a/src/renderer/src/components/dashboard-popout/preview-terminal-options.ts b/src/renderer/src/components/dashboard-popout/preview-terminal-options.ts index 284f6510558..14fe851a657 100644 --- a/src/renderer/src/components/dashboard-popout/preview-terminal-options.ts +++ b/src/renderer/src/components/dashboard-popout/preview-terminal-options.ts @@ -80,7 +80,8 @@ export function buildPreviewTerminalOptions(args: { theme: args.theme ?? undefined, minimumContrastRatio: resolveTerminalMinimumContrastRatio( args.theme?.background, - args.themeMode + args.themeMode, + args.settings?.terminalMinimumContrastRatio ) } } diff --git a/src/renderer/src/components/github-project/group-sort.test.ts b/src/renderer/src/components/github-project/group-sort.test.ts index 57b96267bae..2b787bff357 100644 --- a/src/renderer/src/components/github-project/group-sort.test.ts +++ b/src/renderer/src/components/github-project/group-sort.test.ts @@ -347,3 +347,86 @@ describe('groupRows', () => { expect(groups.map((g) => g.key)).toEqual(['opt_a', '__empty__']) }) }) + +it('indexes field ordering once for grouping and sorting a large project table', () => { + let reads = 0 + const field: GitHubProjectField = { + kind: 'single-select', + id: 'field', + name: 'Status', + dataType: 'SINGLE_SELECT', + options: Array.from({ length: 1000 }, (_, i) => ({ + get id() { + reads++ + return `option-${i}` + }, + name: String(i), + color: 'GRAY' + })) + } + const rows = Array.from({ length: 1000 }, (_, i) => + makeRow(String(i), i, { + field: { + kind: 'single-select', + fieldId: 'field', + optionId: `option-${(i * 173) % 1000}`, + name: String((i * 173) % 1000), + color: 'GRAY' + } + }) + ) + const view = { ...makeView(field, { field, direction: 'ASC' }), groupByFields: [field] } + const table = makeTable(view, rows) + const sorted = sortRows(table, rows) + expect(reads).toBe(1000) + expect( + sorted.map((row) => + Number( + row.fieldValuesByFieldId.field.kind === 'single-select' && + row.fieldValuesByFieldId.field.name + ) + ) + ).toEqual(Array.from({ length: 1000 }, (_, i) => i)) + reads = 0 + const groups = groupRows(table, rows) + expect(reads).toBe(1000) + expect(groups.map((group) => group.key)).toEqual( + Array.from({ length: 1000 }, (_, i) => `option-${i}`) + ) +}) + +it('uses the first iteration ordering and metadata when legacy field IDs repeat', () => { + const field: GitHubProjectField = { + kind: 'iteration', + id: 'iteration', + name: 'Iteration', + dataType: 'ITERATION', + iterations: [ + { id: 'a', title: 'First', startDate: '2026-01-01', duration: 7, completed: true }, + { id: 'b', title: 'Second', startDate: '2026-02-01', duration: 14, completed: false }, + { id: 'a', title: 'Duplicate', startDate: '2026-03-01', duration: 21, completed: false } + ] + } + const rows = ['b', 'a'].map((id, index) => + makeRow(id, index, { + iteration: { + kind: 'iteration', + fieldId: 'iteration', + iterationId: id, + title: id, + startDate: '2026-01-01', + duration: 7 + } + }) + ) + const table = makeTable( + { ...makeView(field, { field, direction: 'ASC' }), groupByFields: [field] }, + rows + ) + expect(sortRows(table, rows).map((row) => row.id)).toEqual(['a', 'b']) + expect(groupRows(table, rows)[0].iteration).toEqual({ + startDate: '2026-01-01', + duration: 7, + completed: true + }) +}) diff --git a/src/renderer/src/components/hover-reveal-touch-action-visibility.test.ts b/src/renderer/src/components/hover-reveal-touch-action-visibility.test.ts index 30c7d53b17b..d47c278582a 100644 --- a/src/renderer/src/components/hover-reveal-touch-action-visibility.test.ts +++ b/src/renderer/src/components/hover-reveal-touch-action-visibility.test.ts @@ -17,6 +17,7 @@ const HOVER_REVEAL_FILES = [ resolve(__dirname, 'editor/DiffSectionHeader.tsx'), resolve(__dirname, 'github-project/ProjectPicker.tsx'), resolve(__dirname, 'github-project/ProjectRow.tsx'), + resolve(__dirname, 'native-chat/NativeChatMessageRow.tsx'), resolve(__dirname, 'right-sidebar/AiVaultSessionRow.tsx'), resolve(__dirname, 'right-sidebar/ChecksPanel.tsx'), resolve(__dirname, 'right-sidebar/local-port-row.tsx'), diff --git a/src/renderer/src/components/jira-issue-sorter.ts b/src/renderer/src/components/jira-issue-sorter.ts index 7fe071dc61e..feaca1bd680 100644 --- a/src/renderer/src/components/jira-issue-sorter.ts +++ b/src/renderer/src/components/jira-issue-sorter.ts @@ -55,6 +55,21 @@ export function sortJiraIssues( orderDirection: JiraIssueSortDirection, jiraPrioritiesBySite: JiraPrioritiesBySite = new Map() ): JiraIssue[] { + const numericKeys = new Map() + if (issues.length > 1 && (orderBy === 'priority' || orderBy === 'updated')) { + for (const issue of issues) { + numericKeys.set( + issue, + orderBy === 'updated' + ? new Date(issue.updatedAt).getTime() + : getJiraPriorityWeight( + issue.priority?.name, + issue.priority?.id, + jiraPrioritiesBySite.get(issue.siteId ?? '') + ) + ) + } + } return [...issues].sort((a, b) => { let comparison = 0 if (orderBy === 'key') { @@ -64,17 +79,13 @@ export function sortJiraIssues( } else if (orderBy === 'status') { comparison = 0 } else if (orderBy === 'priority') { - const prioritiesA = jiraPrioritiesBySite.get(a.siteId ?? '') - const prioritiesB = jiraPrioritiesBySite.get(b.siteId ?? '') - const weightA = getJiraPriorityWeight(a.priority?.name, a.priority?.id, prioritiesA) - const weightB = getJiraPriorityWeight(b.priority?.name, b.priority?.id, prioritiesB) - comparison = weightA - weightB + comparison = numericKeys.get(a)! - numericKeys.get(b)! } else if (orderBy === 'assignee') { const userA = a.assignee?.displayName ?? '' const userB = b.assignee?.displayName ?? '' comparison = userA.localeCompare(userB) } else if (orderBy === 'updated') { - comparison = new Date(a.updatedAt).getTime() - new Date(b.updatedAt).getTime() + comparison = numericKeys.get(a)! - numericKeys.get(b)! } return orderDirection === 'asc' ? comparison : -comparison }) diff --git a/src/renderer/src/components/linear-issue-attribute-filter-primary-team.test.ts b/src/renderer/src/components/linear-issue-attribute-filter-primary-team.test.ts index 91532889d37..5a4bfe471a2 100644 --- a/src/renderer/src/components/linear-issue-attribute-filter-primary-team.test.ts +++ b/src/renderer/src/components/linear-issue-attribute-filter-primary-team.test.ts @@ -36,3 +36,32 @@ describe('resolveLinearIssueAttributeFilterPrimaryTeam', () => { ).toBe('t-a') }) }) + +it('selects a primary team without pairwise membership checks or sorting all teams', () => { + let reads = 0 + let nameReads = 0 + const availableTeams = Array.from({ length: 1000 }, (_, index) => ({ + id: `team-${index}`, + key: String(index), + get name() { + nameReads += 1 + return String((index * 173) % 1000).padStart(4, '0') + } + })) + const selectedTeamIds = new Proxy( + availableTeams.map((team) => team.id), + { + get(target, key, receiver) { + if (typeof key === 'string' && /^\d+$/.test(key)) { + reads += 1 + } + return Reflect.get(target, key, receiver) + } + } + ) + expect(resolveLinearIssueAttributeFilterPrimaryTeam({ selectedTeamIds, availableTeams })).toBe( + availableTeams[0] + ) + expect(reads).toBeLessThanOrEqual(1000) + expect(nameReads).toBeLessThan(5000) +}) diff --git a/src/renderer/src/components/linear-issue-attribute-filter-primary-team.ts b/src/renderer/src/components/linear-issue-attribute-filter-primary-team.ts index 75264f69d49..5797c12a13f 100644 --- a/src/renderer/src/components/linear-issue-attribute-filter-primary-team.ts +++ b/src/renderer/src/components/linear-issue-attribute-filter-primary-team.ts @@ -14,17 +14,19 @@ export function resolveLinearIssueAttributeFilterPrimaryTeam(options: { availableTeams: LinearTeam[] }): LinearTeam | null { const { selectedTeamIds, availableTeams } = options - if (availableTeams.length === 0) { - return null + const selectedIds = new Set(selectedTeamIds) + let firstAvailable: LinearTeam | null = null + let firstSelected: LinearTeam | null = null + for (const team of availableTeams) { + if (!firstAvailable || compareTeamNameId(team, firstAvailable) < 0) { + firstAvailable = team + } + if ( + selectedIds.has(team.id) && + (!firstSelected || compareTeamNameId(team, firstSelected) < 0) + ) { + firstSelected = team + } } - if (selectedTeamIds.length === 0) { - return [...availableTeams].sort(compareTeamNameId)[0] ?? null - } - const selected = availableTeams - .filter((team) => selectedTeamIds.includes(team.id)) - .sort(compareTeamNameId) - if (selected.length > 0) { - return selected[0] ?? null - } - return [...availableTeams].sort(compareTeamNameId)[0] ?? null + return firstSelected ?? firstAvailable } diff --git a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx new file mode 100644 index 00000000000..e137220dd66 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx @@ -0,0 +1,55 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' +import { NativeChatBackgroundTasksStatus } from './NativeChatBackgroundTasksStatus' + +afterEach(cleanup) + +const TASKS: AgentSessionBackgroundTask[] = [ + { id: 'codex-agent:child-1', kind: 'agent', description: 'count_a' }, + { id: 'codex-command:exec-1', kind: 'command', description: 'sleep 90' } +] + +function renderStrip(props: { supportsTaskStop: boolean; supportsStopAll: boolean }): { + onStop: ReturnType +} { + const onStop = vi.fn() + render( + + ) + fireEvent.click(screen.getByRole('button', { expanded: false })) + return { onStop } +} + +describe('NativeChatBackgroundTasksStatus stop affordances', () => { + it('offers a per-task stop on a host that accepts targeted stops', () => { + renderStrip({ supportsTaskStop: true, supportsStopAll: true }) + expect(screen.getByLabelText('Stop count_a')).toBeInTheDocument() + expect(screen.queryByLabelText('Stop background tasks')).not.toBeInTheDocument() + }) + + it('falls back to a stop-all on a host that only accepts an untargeted stop', () => { + renderStrip({ supportsTaskStop: false, supportsStopAll: true }) + expect(screen.getByLabelText('Stop background tasks')).toBeInTheDocument() + }) + + it('offers no stop at all when the provider exposes none', () => { + // Codex: a Stop button here would be a control that cannot act. + renderStrip({ supportsTaskStop: false, supportsStopAll: false }) + expect(screen.queryByLabelText('Stop background tasks')).not.toBeInTheDocument() + expect(screen.queryByLabelText('Stop count_a')).not.toBeInTheDocument() + expect(screen.getByText('count_a')).toBeInTheDocument() + expect(screen.getByText('sleep 90')).toBeInTheDocument() + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx index 9f8f08e1efd..c52b73842a9 100644 --- a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx @@ -26,6 +26,9 @@ function backgroundTaskLabel(task: AgentSessionBackgroundTask): string { export function NativeChatBackgroundTasksStatus(props: { tasks: readonly AgentSessionBackgroundTask[] supportsTaskStop: boolean + /** False when the provider exposes no honest stop at all; the fallback + * control is hidden rather than offering a button that cannot act. */ + supportsStopAll: boolean stoppingTaskIds: ReadonlySet stoppingAll: boolean onStop: (taskId?: string) => void @@ -49,7 +52,7 @@ export function NativeChatBackgroundTasksStatus(props: { - + {translate( 'components.native-chat.backgroundTasks.monitoring', 'Monitoring background tasks' @@ -115,7 +118,7 @@ export function NativeChatBackgroundTasksStatus(props: { )}

)} - {!props.supportsTaskStop ? ( + {!props.supportsTaskStop && props.supportsStopAll ? (
0 ? 'mt-2 border-t border-border pt-2' : 'mt-2'}>
) : null} -